[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-exposes-gaps-in-ai-scientific-reasoning":10,"sections":34},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":24,"tags":25,"sources":29,"feedback":33,"feedback_at":22,"cost_usd":33,"total_tokens":33},8107,"new-benchmark-exposes-gaps-in-ai-scientific-reasoning","New Benchmark Exposes Gaps in AI Scientific Reasoning","SciR tests deduction, induction, and causal reasoning in six LLMs and finds accuracy drops as extraction or inference difficulty rises.","A new benchmark called SciR asks a blunt question: can large language models actually reason through science, or are they just pattern-matching text that looks scientific?\n\nResearchers built SciR around three types of inference that show up constantly in science: deduction, induction, and causal abduction. Rather than relying on human-annotated scientific papers, which are expensive and hard to verify, or synthetic logic puzzles, which look nothing like real research writing, the team generated tasks from formal structures such as deduction trees, inductive rules, and causal graphs. Those structures were then rendered into realistic, multi-document scientific prose using domain-specific genres. The setup lets researchers independently control two separate difficulties: how hard it is to extract the relevant information from the text, and how hard the actual inference is once that information is found.\n\nSix models were tested, and both difficulty axes hurt every one of them, with the effects compounding when stacked together. That is a more nuanced failure mode than most benchmarks capture. Reasoning models like DeepSeek-R1 outperformed plain instruct models specifically on the inference axis, while extraction difficulty turned out to be a separate weakness that existing science benchmarks rarely isolate. Even neurosymbolic pipelines, which hand off the actual logic to a verified solver, still stumbled when the surrounding text got harder to parse, meaning bad prose can sabotage even provably correct reasoning engines.\n\nIt is a useful corrective to \"reasoning model\" marketing, which rarely specifies what kind of reasoning improved, or how much of the reported difficulty was really just finding the right sentence.","[\"ai\",\"llms\",\"benchmarks\",\"research\"]","2026-09-28T04:00:00.000Z","2026-09-28T09:48:04.325Z","2026-09-28T09:48:10.718Z","published",null,[],"ai",[24,26,27,28],"llms","benchmarks","research",[30],{"name":31,"url":32},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2606.13020",0,{"sections":35},[36,39,43,48,53,58,62,67,72,77,82,87,91,96],{"name":37,"slug":24,"count":38,"latest_published_at":18},"AI",4791,{"name":40,"slug":41,"count":42,"latest_published_at":18},"Security","security",762,{"name":44,"slug":45,"count":46,"latest_published_at":47},"Policy","policy",399,"2026-09-27T18:39:02.000Z",{"name":49,"slug":50,"count":51,"latest_published_at":52},"Deals","deals",261,"2026-09-27T15:30:35.000Z",{"name":54,"slug":55,"count":56,"latest_published_at":57},"Hardware","hardware",188,"2026-09-27T20:46:36.000Z",{"name":59,"slug":60,"count":61,"latest_published_at":18},"Science","science",151,{"name":63,"slug":64,"count":65,"latest_published_at":66},"Consumer Tech","consumer-tech",135,"2026-09-26T14:30:00.000Z",{"name":68,"slug":69,"count":70,"latest_published_at":71},"Software","software",91,"2026-09-25T20:55:00.000Z",{"name":73,"slug":74,"count":75,"latest_published_at":76},"Dev Tools","dev-tools",84,"2026-09-26T04:20:58.000Z",{"name":78,"slug":79,"count":80,"latest_published_at":81},"Startups","startups",76,"2026-09-25T18:33:59.000Z",{"name":83,"slug":84,"count":85,"latest_published_at":86},"Gaming","gaming",48,"2026-09-25T18:35:21.000Z",{"name":88,"slug":89,"count":85,"latest_published_at":90},"General","general","2026-09-26T17:02:42.000Z",{"name":92,"slug":93,"count":94,"latest_published_at":95},"Reviews","reviews",30,"2026-09-24T20:07:31.000Z",{"name":97,"slug":98,"count":99,"latest_published_at":100},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]