{"id":"W4409451928","doi":"10.1016/j.jbi.2025.104825","title":"Benchmarking domain-specific pretrained language models to identify the best model for methodological rigor in clinical studies","year":2025,"lang":"en","type":"review","venue":"Journal of Biomedical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University; Hamilton Health Sciences","funders":"EBSCO; Alliance de recherche numérique du Canada; Mitacs","keywords":"Benchmarking; Computer science; Natural language processing; Domain (mathematical analysis); Artificial intelligence; Language model; Data science; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1217259,0.002793567,0.006916852,0.005914738,0.000635124,0.005840575,0.00597812,0.002838507,0.004002146],"category_scores_gemma":[0.2330476,0.001135728,0.008141792,0.004346843,0.001643884,0.006178201,0.003363378,0.004577096,0.001295089],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004441783,"about_ca_system_score_gemma":0.01249324,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005860436,"about_ca_topic_score_gemma":0.009508307,"domain_scores_codex":[0.9431384,0.03929381,0.008965326,0.00298449,0.00517616,0.0004417103],"domain_scores_gemma":[0.7555542,0.2160267,0.009483385,0.007409838,0.01081131,0.0007145305],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001019964,0.0001971919,0.009964114,0.08389983,0.01800801,0.0001479307,0.0006814787,0.02043678,0.00108298,0.01267661,0.01204952,0.8398356],"study_design_scores_gemma":[0.002650904,0.001918455,0.0423889,0.3759709,0.1113271,0.001950721,0.001552597,0.1281677,0.009606034,0.1490029,0.174575,0.0008888309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.01299752,0.8030868,0.1601897,0.009905898,0.001199613,0.002679026,0.004350091,0.001191439,0.004399763],"genre_scores_gemma":[0.2174172,0.4447918,0.3096048,0.007278278,0.001038539,0.006317685,0.01138497,0.0007149262,0.00145188],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8782741,"threshold_uncertainty_score":0.643756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.741286052634215,"score_gpt":0.6411235352893042,"score_spread":0.1001625173449108,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}