[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-framework-helps-ai-agents-write-fast-gpu-code":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},6468,"new-framework-helps-ai-agents-write-fast-gpu-code","New Framework Helps AI Agents Write Fast GPU Code","Ave adds compiler-level guardrails so coding agents stop guessing why their GPU kernels are slow, closing most of the gap with hand-tuned libraries.","AI coding agents can write GPU kernels that work, but not ones that fly.\n\nA new framework called Ave gives those agents a way to check their own optimization ideas before running anything. AI models can already generate correct GPU kernels, but they routinely miss the low-level tricks - shared-memory staging, software pipelining, instruction scheduling - that hand-written expert libraries use to hit full speed. Ave uses a Python-like language where code carries tags describing how data is supposed to flow, then an SMT solver checks those tags against the rules and hands back a concrete counterexample when a proposed change would break something. A planner proposes optimizations from a curated playbook, and a separate agent writes the actual code.\n\nTested on AMD's MI300X chips on matrix multiplication, flash attention, and mixture-of-experts routing - the three kernel types that eat up to 90% of GPU time during LLM inference - Ave-optimized kernels hit 89 to 99% of the throughput of hand-tuned vendor libraries. That beat unguided agentic baselines by anywhere from 1.62x to 1176x, a spread wide enough to suggest some of those baselines were not just slow but broken.\n\nClosing the last few points against libraries AMD's own engineers spent years tuning is a real result, not a marketing number. But a 1176x speedup mostly tells you the starting point was a kernel that barely worked at all.","[\"ai\",\"gpu-optimization\",\"dev-tools\",\"llm-inference\"]","2026-09-16T04:00:00.000Z","2026-09-17T18:06:43.337Z","2026-09-17T18:06:55.247Z","published",null,[],"ai",[24,26,27,28],"gpu-optimization","dev-tools","llm-inference",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.18616",0,{"sections":35},[36,40,45,50,55,59,63,68,73,76,81,86,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":39},"AI",3852,"2026-09-17T08:27:09.000Z",{"name":41,"slug":42,"count":43,"latest_published_at":44},"Security","security",648,"2026-09-17T04:00:00.000Z",{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":44},"Hardware","hardware",154,{"name":60,"slug":61,"count":62,"latest_published_at":44},"Science","science",114,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":74,"slug":27,"count":75,"latest_published_at":44},"Dev Tools",73,{"name":77,"slug":78,"count":79,"latest_published_at":80},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":82,"slug":83,"count":84,"latest_published_at":85},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":87,"slug":88,"count":89,"latest_published_at":90},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]