{"id":"W4399668246","doi":"10.1145/3661167.3661172","title":"The Promise and Challenges of Using LLMs to Accelerate the Screening Process of Systematic Reviews","year":2024,"lang":"en","type":"article","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"Strategic Research Council; Killam Trusts","keywords":"Process (computing); Systematic review; Computer science; Risk analysis (engineering); Medicine; MEDLINE; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007579432,0.00007018478,0.0001770127,0.00001328061,0.00004901758,0.00002057934,0.0001571647,0.00005021251,0.000001232659],"category_scores_gemma":[0.0003855534,0.00002847672,0.00004036482,0.00006931434,0.0001109299,0.000001283763,0.00007361388,0.00003840774,4.840019e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000001370483,"about_ca_system_score_gemma":0.00002201247,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000005606676,"about_ca_topic_score_gemma":0.00001305073,"domain_scores_codex":[0.9993708,0.00008429978,0.0002460646,0.0001319931,0.00007568354,0.00009122752],"domain_scores_gemma":[0.9996082,0.00007815364,0.00006161642,0.0001872086,0.00004048651,0.00002433936],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001576327,0.00007611762,0.0002293939,0.1731894,0.0006054179,0.000004677974,0.005566278,0.00005040923,0.4387287,0.002915593,0.001821127,0.3766552],"study_design_scores_gemma":[0.00112722,0.002427437,0.001232332,0.07057839,0.0008103393,0.0002358006,0.03702509,0.01571668,0.7308106,0.001307145,0.1374694,0.001259495],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.4777288,0.5121603,0.004396467,0.003505136,0.0001330069,0.001332051,0.00000391803,0.00002036004,0.0007199589],"genre_scores_gemma":[0.9878329,0.01048749,0.001369567,0.00005326299,0.00004273376,0.00004982332,5.864017e-7,0.000007115023,0.0001565267],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5101041,"threshold_uncertainty_score":0.1161247,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1590595772908095,"score_gpt":0.3678787959978982,"score_spread":0.2088192187070887,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}