{"id":"W4399327517","doi":"10.1101/2024.06.03.24308405","title":"Evaluating the Efficacy of Large Language Models for Systematic Review and Meta-Analysis Screening","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"","keywords":"Meta-analysis; Systematic review; Computer science; Natural language processing; Psychology; Medicine; MEDLINE; Political science; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006615362,0.0002446556,0.001939415,0.0001555005,0.00007342337,0.0001477914,0.001292142,0.00007817711,0.000009754594],"category_scores_gemma":[0.0006313736,0.0001379446,0.001216687,0.0004009159,0.00001749303,0.00006277704,0.001989332,0.0003160429,0.000001885187],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001331877,"about_ca_system_score_gemma":0.00004976812,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002901797,"about_ca_topic_score_gemma":0.000008471878,"domain_scores_codex":[0.9973223,0.0004552784,0.000847114,0.0006768354,0.0004851316,0.0002132935],"domain_scores_gemma":[0.9966362,0.0008309595,0.0004917269,0.001849611,0.0001477476,0.00004383063],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002200531,0.00003612282,0.000007226601,0.7295818,0.1191398,0.000006573418,0.005896834,0.1110971,0.00003523544,0.03383002,0.00003013222,0.0003369695],"study_design_scores_gemma":[0.00004195008,0.000007669657,0.000001779638,0.002574441,0.242208,0.000001423338,0.00002071769,0.7520005,0.00001306837,0.003028476,4.383199e-7,0.0001015274],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002517772,0.1812384,0.8126801,0.001309655,0.000048929,0.002064129,0.00003138795,0.00006613464,0.00004350197],"genre_scores_gemma":[0.7841881,0.0002733954,0.2126979,0.0008647759,0.00005571765,0.001234985,0.00001265587,0.00003794883,0.0006345171],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7816703,"threshold_uncertainty_score":0.5625218,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1946645548582948,"score_gpt":0.412796460961748,"score_spread":0.2181319061034532,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}