[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"branding":3,"analytics":7,"article-apter-framework-grounds-ai-fine-tuning-in-expert-rubrics":10,"sections":40},{"siteName":4,"siteTagline":5,"publisherName":4,"contactEmail":6},"The Revision","Tech news, decoded.","editor@therevision.news",{"gaMeasurementId":8,"adsenseClientId":9},"G-ZW2MV82GYR","ca-pub-8533917693782264",{"article":11},{"id":12,"slug":13,"title":14,"dek":15,"body_md":16,"tags_json":17,"published_at":18,"created_at":19,"updated_at":20,"status":21,"review_note":22,"review_notes":23,"image_url":22,"persona_id":22,"persona_name":22,"section":30,"tags":31,"sources":35,"feedback":39,"feedback_at":22,"cost_usd":39,"total_tokens":39},5027,"apter-framework-grounds-ai-fine-tuning-in-expert-rubrics","APTER Framework Grounds AI Fine-Tuning In Expert Rubrics","A new post-training framework uses expert-written rubrics instead of one-off grading criteria to fix specific reasoning gaps in math and medical AI models.","Researchers have built a training framework that grades AI models against rubrics written by human experts, then uses the failures to target retraining.\n\nThe system, called APTER, starts from a set of criteria compiled by domain experts, each representing a specific professional skill a model needs to demonstrate. For each question a model answers, APTER pulls the relevant criteria and turns them into a query-specific checklist that can be scored without a reference answer. Those scores double as a diagnostic: when the same criterion keeps scoring low across many examples, the framework flags it as a persistent weakness and triggers targeted supervised fine-tuning during reinforcement learning, instead of retraining on everything at once. The team tested this on mathematical reasoning and medical question answering across three generations of models.\n\nThis matters because most reward models grade for fluency and surface correctness without telling you which specific skill a model is missing. Tying evaluation to stable, named criteria instead of ad hoc per-query rubrics means a lab can actually see where a model keeps failing and fix that, rather than guessing. The reported gains were substantial: up to 15.86 points on math averages and 8.04 points on medical averages over base models.\n\nAPTER is a training method for closing specific skill gaps, not a public benchmark, and the strongest evidence so far comes from the same team that built it; independent replication on domains beyond math and medicine would settle whether the gains generalize.","[\"ai\",\"fine-tuning\",\"llm training\",\"reasoning\"]","2026-08-17T04:00:00.000Z","2026-08-17T06:20:12.406Z","2026-08-17T06:20:24.247Z","published",null,[24],{"id":25,"reviewer":26,"round":27,"reason":28,"status":29},"editor-r1","editor",1,"The closing line calls APTER 'the benchmark' built by the same team, but the article (correctly, per source) describes APTER throughout as a post-training\u002Ffine-tuning framework, not a benchmark — fix this inconsistent\u002Funsupported characterization before publishing.","resolved","ai",[30,32,33,34],"fine-tuning","llm training","reasoning",[36],{"name":37,"url":38},"arXiv cs.AI","https:\u002F\u002Farxiv.org\u002Fabs\u002F2608.14212",0,{"sections":41},[42,46,50,55,60,65,70,75,80,85,90,95,100,105],{"name":43,"slug":30,"count":44,"latest_published_at":45},"AI",3293,"2026-08-20T04:00:00.000Z",{"name":47,"slug":48,"count":49,"latest_published_at":45},"Security","security",435,{"name":51,"slug":52,"count":53,"latest_published_at":54},"Policy","policy",210,"2026-08-19T09:32:27.000Z",{"name":56,"slug":57,"count":58,"latest_published_at":59},"Deals","deals",179,"2026-06-29T20:02:07.000Z",{"name":61,"slug":62,"count":63,"latest_published_at":64},"Hardware","hardware",140,"2026-08-19T18:25:42.000Z",{"name":66,"slug":67,"count":68,"latest_published_at":69},"Consumer Tech","consumer-tech",95,"2026-08-18T16:05:00.000Z",{"name":71,"slug":72,"count":73,"latest_published_at":74},"Science","science",90,"2026-08-19T18:41:02.000Z",{"name":76,"slug":77,"count":78,"latest_published_at":79},"Software","software",73,"2026-08-18T07:51:50.000Z",{"name":81,"slug":82,"count":83,"latest_published_at":84},"Dev Tools","dev-tools",69,"2026-08-18T04:00:00.000Z",{"name":86,"slug":87,"count":88,"latest_published_at":89},"Startups","startups",47,"2026-08-19T19:13:46.000Z",{"name":91,"slug":92,"count":93,"latest_published_at":94},"Gaming","gaming",41,"2026-07-09T04:00:00.000Z",{"name":96,"slug":97,"count":98,"latest_published_at":99},"General","general",33,"2026-08-18T22:18:13.000Z",{"name":101,"slug":102,"count":103,"latest_published_at":104},"Reviews","reviews",20,"2026-06-24T12:00:01.000Z",{"name":106,"slug":107,"count":108,"latest_published_at":109},"How-To","how-to",6,"2026-06-16T09:00:00.000Z"]