{"id":"W4399280155","doi":"10.1101/2024.06.01.24308323","title":"Prompting is all you need: LLMs for systematic review screening","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Sciences Centre; University of Calgary; Sunnybrook Health Science Centre; Vector Institute; South Health Campus; University of Toronto","funders":"","keywords":"Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4379441,0.003040582,0.004894464,0.01125869,0.002482149,0.01222599,0.004624364,0.005720995,0.02538022],"category_scores_gemma":[0.8443523,0.003684967,0.00623192,0.01150403,0.004093334,0.02008839,0.01407492,0.005600893,0.008617919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007373751,"about_ca_system_score_gemma":0.03755176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002366761,"about_ca_topic_score_gemma":0.007487588,"domain_scores_codex":[0.3917862,0.516087,0.05530092,0.01391609,0.02071851,0.002191153],"domain_scores_gemma":[0.09058069,0.7736052,0.04357546,0.05911082,0.02935221,0.003775598],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004783963,0.0003965874,0.009733003,0.05069596,0.002024964,0.0005800145,0.01409541,0.004611711,0.004115082,0.02544744,0.1219213,0.7615947],"study_design_scores_gemma":[0.01087003,0.002894094,0.01302114,0.05984986,0.004830047,0.001879769,0.008392203,0.08501205,0.01247291,0.4600494,0.3387523,0.001976303],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02646096,0.0127374,0.7388482,0.1355401,0.005717827,0.02489879,0.008993933,0.03547278,0.01133007],"genre_scores_gemma":[0.07700443,0.00125583,0.8963362,0.009266944,0.0008724321,0.01249545,0.001256474,0.0009205067,0.0005917897],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5620559,"threshold_uncertainty_score":0.6931151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.741785538676171,"score_gpt":0.5428035855572323,"score_spread":0.1989819531189387,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}