[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-new-benchmark-shows-llms-still-fumble-payment-rules":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},6517,"new-benchmark-shows-llms-still-fumble-payment-rules","New Benchmark Shows LLMs Still Fumble Payment Rules","BenchCompass tests 16 models on payment-domain reasoning and finds even top performers stumble when inputs turn adversarial.","A new benchmark says the AI models banks might eventually lean on for payment operations still don't reliably know the rules.\n\nResearchers behind BENCHCOMPASS built a payment-domain test suite from scenario-grounded evidence packs, then ran LLM-based quality checks, added adversarial variants of the task inputs, and had domain experts sign off on the final question set. The result is two tiers: an expert-reviewed \"Pro\" benchmark covering payment knowledge, context-grounded reasoning, and robustness under attacked inputs, plus a looser \"Normal\" pool held back for future curation. The team ran 16 model variants through it. The best frontier model scored 89.6% on open context-grounded reasoning but dropped to 81.7% once inputs were adversarially perturbed, while a representative 32B open-weight model fell from 69.8% to 42.6% under the same attack.\n\nThat gap matters because it separates three failure modes that generic benchmarks tend to blur together: models that don't know payment rules, models that know the rules but reason poorly over the evidence they're handed, and models that fold when inputs get messy or misleading. In a domain where the correct answer depends on transaction state, participant role, region, and payment rail, that last failure mode is the one that actually breaks things in production.\n\nThe widening gap between clean and attacked scores, especially for smaller open-weight models, is the kind of detail that tends to get left out of vendor pitch decks promising payments-ready AI.","[\"ai\",\"benchmarks\",\"payments\",\"llm-evaluation\"]","2026-09-17T04:00:00.000Z","2026-09-17T21:58:01.091Z","2026-09-17T21:58:12.945Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"Drop or explicitly frame as speculation the closing claim that 'the payments industry is quietly piloting LLM copilots for compliance and reconciliation' — this isn't supported anywhere in the source material and reads as an asserted fact rather than analysis.","resolved","ai",[30,32,33,34],"benchmarks","payments","llm-evaluation",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2609.18270",0,{"sections":41},[42,46,50,55,60,64,68,73,78,82,87,92,97,102],{"name":43,"slug":30,"count":44,"latest_published_at":45},"AI",3852,"2026-09-17T08:27:09.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":18},"Security","security",648,{"name":51,"slug":52,"count":53,"latest_published_at":54},"Policy","policy",338,"2026-09-11T04:00:00.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":18},"Hardware","hardware",154,{"name":65,"slug":66,"count":67,"latest_published_at":18},"Science","science",114,{"name":69,"slug":70,"count":71,"latest_published_at":72},"Consumer Tech","consumer-tech",99,"2026-09-09T17:27:33.000Z",{"name":74,"slug":75,"count":76,"latest_published_at":77},"Software","software",75,"2026-09-10T20:41:21.000Z",{"name":79,"slug":80,"count":81,"latest_published_at":18},"Dev Tools","dev-tools",73,{"name":83,"slug":84,"count":85,"latest_published_at":86},"Startups","startups",55,"2026-09-09T23:14:29.000Z",{"name":88,"slug":89,"count":90,"latest_published_at":91},"Gaming","gaming",43,"2026-09-10T12:18:06.000Z",{"name":93,"slug":94,"count":95,"latest_published_at":96},"General","general",41,"2026-09-08T01:57:23.000Z",{"name":98,"slug":99,"count":100,"latest_published_at":101},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":103,"slug":104,"count":105,"latest_published_at":106},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]