[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-small-open-source-llms-learn-to-hack-almost-as-well-as-gpt-4o":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},4927,"small-open-source-llms-learn-to-hack-almost-as-well-as-gpt-4o","Small Open-Source LLMs Learn to Hack Almost as Well as GPT-4o","New prompting and reflection techniques push small open-weight LLMs from 8% to 67% success at Linux privilege escalation, rivaling GPT-4o.","Researchers just closed most of the gap between small, locally run AI models and the cloud giants at a very specific skill: breaking into Linux systems.\n\nThe study took the open-source hackingBuddyGPT framework and identified six recurring reasons small language models fail at automated Linux privilege escalation. Researchers then applied five known techniques - chain-of-thought prompting, retrieval-augmented generation, structured prompting, history compression, and reflective analysis - to two small open-weight models, Llama3.1 8B and Qwen2.5 7B. Under identical test conditions, those models jumped from an 8% success rate to 67%, matching a guided GPT-4o. A larger open-weight model, Llama3.1 70B, hit 83%. The biggest single contributor was reflection, where the model reviews and critiques its own attempted steps, and the researchers found the real bottleneck for small models was spotting vulnerabilities in the first place, not exploiting them once found.\n\nThat distinction matters beyond the benchmark. Security teams have been wary of sending internal system data to cloud LLMs for penetration testing, citing privacy and sovereignty concerns. This work suggests you no longer need a frontier cloud model to automate that work - a properly scaffolded local model, running entirely on hardware you control, gets most of the way there.\n\nThe paper pitches these findings as lessons for defenders. They read just as easily as a recipe for anyone building an attacker.","[\"ai agents\",\"llm security\",\"penetration testing\",\"open source\"]","2026-08-13T04:00:00.000Z","2026-08-14T17:55:31.712Z","2026-08-14T17:55:43.502Z","published",null,[],"security",[26,27,28,29],"ai agents","llm security","penetration testing","open source",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2604.27143",0,{"sections":36},[37,42,45,50,55,60,65,70,75,80,85,90,95,100],{"name":38,"slug":39,"count":40,"latest_published_at":41},"AI","ai",3293,"2026-08-20T04:00:00.000Z",{"name":43,"slug":24,"count":44,"latest_published_at":41},"Security",435,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]