{"id":"W4412432233","doi":"10.1016/j.jval.2025.04.1205","title":"MSR53 Development and Validation of an LLM-Based Study Selection Tool for Automating Systematic Literature Reviews: Achieving Time Efficiency and Maintaining Gold Standard Accuracy","year":2025,"lang":"en","type":"article","venue":"Value in Health","topic":"Impact of AI and Big Data on Business and Society","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"EVERSANA (Canada)","funders":"","keywords":"Selection (genetic algorithm); Gold standard (test); Systematic review; Computer science; Systematic error; Machine learning; Statistics; MEDLINE; Mathematics; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1965897,0.002092596,0.003735017,0.02629142,0.002808387,0.009427419,0.003458432,0.00300212,0.01620174],"category_scores_gemma":[0.5041247,0.002411405,0.007839301,0.01253232,0.001211909,0.004385844,0.007187678,0.002123463,0.005635524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003682037,"about_ca_system_score_gemma":0.02892616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004370697,"about_ca_topic_score_gemma":0.01410919,"domain_scores_codex":[0.8292732,0.07253134,0.06774985,0.00904066,0.02002382,0.001381093],"domain_scores_gemma":[0.3850136,0.4454851,0.04366643,0.03399732,0.08838012,0.003457382],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002603058,0.0004726667,0.03959124,0.06352129,0.007766444,0.0006546475,0.005332889,0.004025986,0.01197143,0.01011658,0.07394652,0.7799973],"study_design_scores_gemma":[0.007825776,0.003259994,0.1093978,0.05677163,0.02206206,0.002959755,0.004113154,0.138652,0.06856978,0.03940609,0.5452231,0.001758886],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04466306,0.01188008,0.7518069,0.01068956,0.001314763,0.04778535,0.05603302,0.06005077,0.01577661],"genre_scores_gemma":[0.03378579,0.0006706929,0.933058,0.0008851787,0.0001270158,0.0216384,0.007266499,0.001051848,0.001516608],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8034103,"threshold_uncertainty_score":0.9907479,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08687134925101475,"score_gpt":0.4070922405465321,"score_spread":0.3202208912955173,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}