[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-tests-ai-at-finding-topics-in-old-czech-texts":10,"sections":35},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":30,"feedback":34,"feedback_at":22,"cost_usd":34,"total_tokens":34},6701,"new-benchmark-tests-ai-at-finding-topics-in-old-czech-texts","New Benchmark Tests AI at Finding Topics in Old Czech Texts","A new human-annotated benchmark shows large language models can spot topics in historical Czech documents but often fail to pinpoint exactly where they appear.","A new benchmark shows AI language models can often tell you a topic exists in an old Czech document, but not always where.\n\nResearchers built CzechTopic, a benchmark drawn from historical Czech documents, where human annotators marked text spans matching topics defined by a name and description. The dataset supports checking results at both the whole-document level and the individual-word level, and instead of grading against one \"correct\" answer, it measures models against how much multiple human annotators agreed with each other. The team tested a range of large language models against smaller BERT-based models that had been fine-tuned on a distilled development set. The dataset and evaluation code are public on GitHub.\n\nThe results split the field. Some LLMs got close to human-level agreement on identifying that a topic was present, but the same models often failed badly at localizing the actual span of text - the harder, more useful task for anyone trying to search an archive. Meanwhile, the fine-tuned BERT models, despite being far smaller, stayed competitive with the best LLMs.\n\nThat gap matters beyond one benchmark. It's a reminder that \"finding the right idea in a document\" and \"finding exactly where that idea lives\" are different skills, and that scale doesn't automatically buy you the second one. For archives and libraries digitizing historical text in smaller languages, a cheap fine-tuned model may still beat a general-purpose LLM on precision. Given the code is on GitHub, that's now a testable claim rather than a marketing line.","[\"ai benchmarks\",\"nlp\",\"historical documents\",\"czech language\"]","2026-09-17T04:00:00.000Z","2026-09-18T06:46:23.993Z","2026-09-18T06:46:35.844Z","published",null,[],"ai",[26,27,28,29],"ai benchmarks","nlp","historical documents","czech language",[31],{"name":32,"url":33},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2603.03884",0,{"sections":36},[37,41,45,50,55,59,63,68,73,77,82,87,92,97],{"name":38,"slug":24,"count":39,"latest_published_at":40},"AI",3853,"2026-09-17T08:27:09.000Z",{"name":42,"slug":43,"count":44,"latest_published_at":18},"Security","security",648,{"name":46,"slug":47,"count":48,"latest_published_at":49},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":51,"slug":52,"count":53,"latest_published_at":54},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":18},"Hardware","hardware",154,{"name":60,"slug":61,"count":62,"latest_published_at":18},"Science","science",114,{"name":64,"slug":65,"count":66,"latest_published_at":67},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":69,"slug":70,"count":71,"latest_published_at":72},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]