{"id":"W4399210824","doi":"10.2196/56780","title":"Potential Roles of Large Language Models in the Production of Systematic Reviews and Meta-Analyses","year":2024,"lang":"en","type":"article","venue":"Journal of Medical Internet Research","topic":"Clinical practice guidelines implementation","field":"Medicine","cited_by":59,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; Impact","funders":"","keywords":"Preprint; Meta-analysis; Production (economics); Computer science; Linguistics; Economics; Philosophy; World Wide Web; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7382747,0.006986426,0.01005225,0.01957216,0.003478076,0.02408715,0.008833555,0.006172112,0.01164779],"category_scores_gemma":[0.9289092,0.008825509,0.01413793,0.02241279,0.01064154,0.02157556,0.01832298,0.01030626,0.003129818],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0111586,"about_ca_system_score_gemma":0.03629173,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00471594,"about_ca_topic_score_gemma":0.01135137,"domain_scores_codex":[0.1283047,0.7812123,0.05686967,0.01383071,0.01897438,0.000808284],"domain_scores_gemma":[0.01582977,0.942335,0.01582541,0.0207082,0.004871876,0.0004296848],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004767798,0.0003193642,0.01944568,0.1372049,0.04344257,0.002554424,0.02093808,0.02664522,0.0021347,0.3325098,0.03435223,0.3756853],"study_design_scores_gemma":[0.002525324,0.0005234969,0.002838484,0.03804154,0.01819731,0.000919995,0.001347934,0.07820832,0.003161814,0.8037146,0.04956101,0.0009601939],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005120323,0.03037248,0.905705,0.03499275,0.003255224,0.006434857,0.004034232,0.003864222,0.006220925],"genre_scores_gemma":[0.08866688,0.006450582,0.8591874,0.009725686,0.001978233,0.0303263,0.001329003,0.001323997,0.001011896],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2617253,"threshold_uncertainty_score":0.322754,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6880497700622092,"score_gpt":0.6641778485082145,"score_spread":0.02387192155399476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}