{"id":"W4395686571","doi":"10.48550/arxiv.2404.15667","title":"The Promise and Challenges of Using LLMs to Accelerate the Screening Process of Systematic Reviews","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Strategic Research Council; Killam Trusts","keywords":"Process (computing); Risk analysis (engineering); Computer science; Business","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2866312,0.003213106,0.003667052,0.01105492,0.001661682,0.008875527,0.003837464,0.003313679,0.00783474],"category_scores_gemma":[0.6664684,0.003273263,0.004683176,0.008713908,0.002012198,0.0146755,0.00682657,0.004431766,0.004013102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005166354,"about_ca_system_score_gemma":0.01877527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00447916,"about_ca_topic_score_gemma":0.01245938,"domain_scores_codex":[0.6830642,0.2763572,0.01900997,0.007131923,0.01346682,0.0009698129],"domain_scores_gemma":[0.0819171,0.8428898,0.01985921,0.03270095,0.02045624,0.002176706],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005503831,0.0005972201,0.007988522,0.04043217,0.002833027,0.0002644629,0.008371516,0.0119056,0.01498354,0.008149719,0.02601614,0.8729542],"study_design_scores_gemma":[0.01398484,0.01189371,0.04475019,0.03662027,0.009430961,0.002833423,0.007773097,0.4040138,0.05317204,0.1882323,0.2245617,0.002733646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09300417,0.03494953,0.7462935,0.0510479,0.001998923,0.01016728,0.004294015,0.05093851,0.007306293],"genre_scores_gemma":[0.07113718,0.002389671,0.919904,0.00186716,0.0002932808,0.002382964,0.0008597406,0.0007249438,0.000441109],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7133688,"threshold_uncertainty_score":0.8797107,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8993163377795477,"score_gpt":0.4303422830797612,"score_spread":0.4689740546997864,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}