[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-magnitude-claims-2x-faster-local-ai-inference-than-llamacpp":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},8807,"magnitude-claims-2x-faster-local-ai-inference-than-llamacpp","Magnitude Claims 2x Faster Local AI Inference Than llama.cpp","The YC-backed startup built its own GPU kernels to speed up local AI agents, though its 2x claim comes from benchmarks it ran itself.","Magnitude, a Y Combinator-backed startup, has released an open-source inference engine that it says runs AI agents up to twice as fast as the go-to tool llama.cpp, on the same hardware.\n\nFounders Anders and Tom, who previously built an open-source browser agent that hit 4,000-plus GitHub stars, say they built Magnitude because no existing engine fit how they wanted to run local AI agents. Their pitch: engines like vLLM and SGLang are built for crunching many requests at once in a data center, while llama.cpp and Ollama trade speed for working on almost any hardware. Magnitude instead tests and tunes its code against your specific chip before a model runs, then only sets aside the memory it needs to hold that model, growing or shrinking as agent sessions start and stop. In benchmarks the company ran itself against llama.cpp on a Qwen 3.6 35B model, Magnitude reported generating text up to 92% faster on a Mac M4 Pro and 19% faster on an Nvidia DGX Spark, while using about 27-28% less memory per agent session on both.\n\nRunning agents locally instead of through a cloud API keeps costs and data off someone else's servers, but it also means juggling long-running sessions on a machine you still want to use for other things. Magnitude is betting that per-device tuning, borrowed from ideas in academic work like FlashAttention and engines like SGLang, can deliver both speed and flexibility instead of forcing a choice between them. Because the code is open source under an Apache 2.0 license, outside developers can actually check those numbers instead of taking the company's word for it.\n\nIt's a narrow, technical bet in a crowded field of inference engines, and the real verdict won't come from a 32-point Hacker News thread but from whether those benchmarks hold up once other people run them.","[\"ai-agents\",\"inference-engines\",\"open-source\",\"yc-startups\"]","2026-09-30T17:37:40.000Z","2026-10-01T04:06:03.457Z","2026-10-01T04:06:08.822Z","published",null,[],"ai",[26,27,28,29],"ai-agents","inference-engines","open-source","yc-startups",[31],{"name":32,"url":33},"Hacker News","https:\u002F\u002Fgithub.com\u002Fmagnitudedev\u002Fmagnitude",0,{"sections":36},[37,41,46,51,56,61,66,71,76,80,85,90,95,100],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",5236,"2026-09-30T20:13:50.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":45},"Security","security",799,"2026-09-30T19:29:08.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",427,"2026-09-30T20:36:54.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",298,"2026-09-30T21:00:26.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",196,"2026-09-30T13:00:00.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Science","science",157,"2026-09-30T15:00:56.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Consumer Tech","consumer-tech",148,"2026-09-30T17:53:35.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Dev Tools","dev-tools",92,"2026-09-30T18:17:45.000Z",{"name":77,"slug":78,"count":74,"latest_published_at":79},"Software","software","2026-09-30T17:45:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",83,"2026-09-29T21:51:36.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",50,"2026-09-30T16:24:30.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",49,"2026-09-28T16:44:57.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",31,"2026-09-28T14:31:34.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]