[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-researchers-train-vision-ai-on-images-ai-itself-made":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},9288,"researchers-train-vision-ai-on-images-ai-itself-made","Researchers Train Vision AI on Images AI Itself Made","VisionFoundry generates synthetic images and questions to fix vision-language models' blind spots, boosting spatial reasoning benchmarks by up to 10.5%.","Vision-language models (VLMs) are great at describing a photo but surprisingly bad at basic spatial questions, like which object is closer to the camera. A new paper proposes fixing that gap by training on images the AI generated for itself.\n\nThe pipeline, called VisionFoundry, starts with nothing but a task name - say, \"viewpoint recognition.\" An LLM writes matching questions, answers, and prompts for a text-to-image model, which then generates the actual pictures. A multimodal filter checks the results and tosses out bad samples before they ever reach training. The team used this process to build VisionFoundry-10k, a synthetic dataset covering 10 distinct perception tasks, and fine-tuned three open-source VLMs on it. The gains were consistent: Qwen2.5-VL-3B-Instruct improved 6.7% on the MMVP-pair benchmark and 10.5% on CV-Bench-3D, and the approach also helped when used as reinforcement-learning data instead of plain fine-tuning.\n\nThis matters because VLMs have been scaling fast on language while still fumbling simple visual logic, mostly because natural photo datasets rarely come labeled with the spatial or viewpoint detail these tasks need. Generating that supervision synthetically, with no manual annotation, is a cheaper and more scalable fix than waiting for better-labeled real-world data to show up.\n\nIt is also the same trick that rescued LLM training once real text got scarce, now aimed at pixels - which is reassuring, but benchmark gains on lab backbones are not the same as proof this holds up in whatever model actually ships to your phone.","[\"vlm\",\"synthetic-data\",\"ai-research\",\"computer-vision\"]","2026-10-01T04:00:00.000Z","2026-10-02T07:10:11.339Z","2026-10-02T07:10:12.934Z","published",null,[],"ai",[26,27,28,29],"vlm","synthetic-data","ai-research","computer-vision",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.09531",0,{"sections":36},[37,40,44,48,53,58,62,67,72,76,81,86,91,96],{"name":38,"slug":24,"count":39,"latest_published_at":18},"AI",5671,{"name":41,"slug":42,"count":43,"latest_published_at":18},"Security","security",820,{"name":45,"slug":46,"count":47,"latest_published_at":18},"Policy","policy",430,{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",165,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",149,"2026-09-30T22:57:11.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Dev Tools","dev-tools",93,"2026-10-01T02:30:48.000Z",{"name":73,"slug":74,"count":70,"latest_published_at":75},"Software","software","2026-09-30T21:41:11.000Z",{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",84,"2026-09-30T20:39:09.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",51,"2026-09-30T16:24:30.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",50,"2026-09-30T21:37:54.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]