[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-llms-ace-colombian-law-quizzes-but-flunk-real-answers":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},9959,"llms-ace-colombian-law-quizzes-but-flunk-real-answers","LLMs Ace Colombian Law Quizzes but Flunk Real Answers","A new benchmark finds AI models answer Colombian legal multiple-choice questions well but get free-text answers wrong more than half the time.","A new benchmark shows top language models can ace Colombian law trivia but still get the actual legal reasoning wrong most of the time.\n\nResearchers built a 1,042-item benchmark spanning ten areas of Colombian law in three formats: multiple-choice, semi-open, and open-ended IRAC answers, all vetted by legal experts through a multi-stage review process. They tested 15 proprietary and open-weight models. On multiple-choice questions, accuracy ranged from 0.905 for Gemini 3.1 Pro down to 0.577 for the weakest model. On free-text answers, the kind a lawyer or citizen would actually ask for, no model topped 0.45 out of 1.0 for factual correctness, and only about half the legal provisions models cited were real and correctly applied.\n\nThe gap between formats is the real finding. Confidence and relevance track separately from correctness, with a negative correlation between how responsive an answer sounds and how right it is. That is a bad combination for non-expert users leaning on a chatbot for legal guidance, exactly the population most likely to try it.\n\nMost LLM legal benchmarks are built around U.S. or EU case law, so this is a useful reminder that fluency in a legal system's language is not the same as knowing its law, a gap that likely widens for jurisdictions with less English-language training data.","[\"ai\",\"llm-evaluation\",\"legal-tech\",\"benchmarks\"]","2026-10-05T04:00:00.000Z","2026-10-05T15:23:59.272Z","2026-10-05T15:24:03.712Z","published",null,[],"ai",[24,26,27,28],"llm-evaluation","legal-tech","benchmarks",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2610.03639",0,{"sections":35},[36,39,43,48,53,58,62,67,71,75,80,85,90,95],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",6170,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",860,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",444,"2026-10-03T15:02:01.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",323,"2026-10-04T13:00:00.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",204,"2026-10-03T14:50:50.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",177,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",158,"2026-10-03T03:21:12.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":18},"Dev Tools","dev-tools",97,{"name":72,"slug":73,"count":70,"latest_published_at":74},"Software","software","2026-10-04T10:00:00.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Startups","startups",92,"2026-10-04T14:36:25.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Gaming","gaming",53,"2026-10-02T02:50:39.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"General","general",51,"2026-10-05T02:35:01.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Reviews","reviews",32,"2026-10-02T18:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"How-To","how-to",7,"2026-10-01T09:00:00.000Z"]