[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-ai-pipeline-pulls-plant-traits-from-old-botany-papers":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},5110,"ai-pipeline-pulls-plant-traits-from-old-botany-papers","AI Pipeline Pulls Plant Traits From Old Botany Papers","Researchers combined OCR, rule-based parsing, and LLM ensembles to pull over 55,000 botanical trait annotations from three regional plant datasets.","Researchers built an agent-based AI pipeline that reads scanned botany papers and automatically tags each plant species with its traits.\n\nThe system starts with optical character recognition to turn PDF botanical descriptions into machine-readable text, then segments and indexes that text by genus and species. Rule-based parsers pull out structured traits like leaf shape or flower color, and ensembles of large language models expand the trait vocabulary and untangle ambiguous wording. Tested on three regional botanical datasets, the pipeline extracted 55,737 trait annotations across 4,961 species, or roughly 11 traits per species. Adding the LLM enrichment step improved coverage for 75% of traits and lifted total annotations by 59% over the rule-based parsers alone.\n\nBotanical knowledge is still locked inside decades of dense, inconsistently formatted PDFs and field guides, and manually tagging that text does not scale. Pairing rigid rule-based extraction with LLMs for the messy edge cases is a workable template for other fields sitting on similar piles of unstructured legacy documents, from medical case reports to geological surveys. The researchers also found that swapping OCR engines barely changed species recognition, which is a decent sign the pipeline is not just tuned to one dataset's quirks.\n\nStill, this is three regional datasets, not a global flora catalog, and the rule-based layer means someone has to keep writing parsing rules as new document formats show up. Calling it \"agentic\" is generous when the core trick is LLMs cleaning up after rule-based parsers, not the models doing the reasoning themselves.","[\"ai\",\"llm agents\",\"botany\",\"data extraction\"]","2026-08-18T04:00:00.000Z","2026-08-18T05:27:02.756Z","2026-08-18T05:27:14.572Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"publisher-r1","publisher",1,"The stated average of \"about nine traits per species\" is inconsistent with the reported totals (55,737 annotations ÷ 4,961 species ≈ 11.2 traits per species, not nine).","resolved","ai",[30,32,33,34],"llm agents","botany","data extraction",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.14587",0,{"sections":41},[42,46,50,55,60,65,70,75,80,84,89,94,99,104],{"name":43,"slug":30,"count":44,"latest_published_at":45},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":45},"Security","security",435,{"name":51,"slug":52,"count":53,"latest_published_at":54},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":18},"Dev Tools","dev-tools",69,{"name":85,"slug":86,"count":87,"latest_published_at":88},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":90,"slug":91,"count":92,"latest_published_at":93},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":95,"slug":96,"count":97,"latest_published_at":98},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":100,"slug":101,"count":102,"latest_published_at":103},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":105,"slug":106,"count":107,"latest_published_at":108},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]