{"id":"W4399327517","doi":"10.1101/2024.06.03.24308405","title":"Evaluating the Efficacy of Large Language Models for Systematic Review and Meta-Analysis Screening","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"","keywords":"Meta-analysis; Systematic review; Computer science; Natural language processing; Psychology; Medicine; MEDLINE; Political science; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5114839,0.004482095,0.006783622,0.01264032,0.002484152,0.009209514,0.005160984,0.004576907,0.01276682],"category_scores_gemma":[0.8218209,0.0038153,0.0201642,0.01311518,0.002958019,0.008915904,0.0083823,0.005169604,0.002607167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005665074,"about_ca_system_score_gemma":0.02627439,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003341627,"about_ca_topic_score_gemma":0.009836741,"domain_scores_codex":[0.4674453,0.4546662,0.0477652,0.013123,0.01578971,0.001210681],"domain_scores_gemma":[0.08626276,0.8467523,0.02338707,0.02770888,0.01493626,0.0009527208],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01636429,0.0009236427,0.02545634,0.1863479,0.03982868,0.001334248,0.006382078,0.1000195,0.003927531,0.03151936,0.05333883,0.5345576],"study_design_scores_gemma":[0.01823894,0.004570292,0.01575656,0.04728443,0.06024392,0.001292537,0.001286779,0.6168072,0.0100137,0.1583071,0.0644216,0.001777048],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03224586,0.01071311,0.8461773,0.01367127,0.001415325,0.05525022,0.01404415,0.02204826,0.004434463],"genre_scores_gemma":[0.08113744,0.001112209,0.8645164,0.001204675,0.0001544301,0.0488867,0.001984127,0.0006657558,0.0003382807],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4885161,"threshold_uncertainty_score":0.6024274,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1946645548582948,"score_gpt":0.412796460961748,"score_spread":0.2181319061034532,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}