[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-study-finds-llms-struggle-to-verify-rust-code-like-provers-do":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},5592,"study-finds-llms-struggle-to-verify-rust-code-like-provers-do","Study Finds LLMs Struggle to Verify Rust Code Like Provers Do","A new benchmark testing whether language models can reason through formal Rust proofs finds they crumble when steps go missing.","A new benchmark says today's AI models are bad at one very specific, very unglamorous skill: filling in gaps in formal proofs that verify Rust code is safe.\n\nResearchers built VCoT-Lift, a tool that converts the low-level reasoning of automated theorem provers into readable, step-by-step logic humans can follow. Using that, they created VCoT-Bench, a set of 1,988 tasks that ask models to complete partial verification proofs for Rust programs. The benchmark scores models on three things: how they handle varying amounts of missing proof, how well they manage different proof types, and whether the location of a gap in the proof trips them up. Ten current models were tested. All ten struggled.\n\nThis matters because Rust's whole pitch is memory safety enforced by the compiler, and formal verification is the next layer up - mathematically proving a program behaves correctly, not just hoping tests catch the bugs. If companies want AI to write or check safety-critical Rust rather than just autocomplete it, the model needs to reason the way a theorem prover does: rigorously, not plausibly. The benchmark shows models are still pattern-matching their way through proofs rather than actually deriving them, and that gap widens as proofs get sparser or trickier.\n\nIt's a reminder that fluent code suggestions and sound formal reasoning are different skills, and right now LLMs have plenty of the former and not much of the latter.","[\"rust\",\"llm-benchmarks\",\"formal-verification\",\"ai-research\"]","2026-08-18T04:00:00.000Z","2026-08-19T02:47:46.387Z","2026-08-19T02:47:58.305Z","published",null,[],"dev-tools",[26,27,28,29],"rust","llm-benchmarks","formal-verification","ai-research",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.18334",0,{"sections":36},[37,42,46,51,56,61,66,71,76,79,84,89,94,99],{"name":38,"slug":39,"count":40,"latest_published_at":41},"AI","ai",3293,"2026-08-20T04:00:00.000Z",{"name":43,"slug":44,"count":45,"latest_published_at":41},"Security","security",435,{"name":47,"slug":48,"count":49,"latest_published_at":50},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":52,"slug":53,"count":54,"latest_published_at":55},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":57,"slug":58,"count":59,"latest_published_at":60},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":62,"slug":63,"count":64,"latest_published_at":65},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":67,"slug":68,"count":69,"latest_published_at":70},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":72,"slug":73,"count":74,"latest_published_at":75},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":77,"slug":24,"count":78,"latest_published_at":18},"Dev Tools",69,{"name":80,"slug":81,"count":82,"latest_published_at":83},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":85,"slug":86,"count":87,"latest_published_at":88},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]