{"id":"W4388380341","doi":"10.31234/osf.io/dc6tz","title":"Evaluating Large Language Models for Assisting in Meta-Analysis","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Natural Science Foundation of China","keywords":"Coding (social sciences); Meta-analysis; Computer science; Qualitative analysis; Perspective (graphical); Empirical research; Qualitative property; Quantitative analysis (chemistry); Natural language processing; Data science; Qualitative research; Psychology; Artificial intelligence; Machine learning; Statistics; Social science; Sociology; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3893289,0.006527099,0.006219397,0.02033028,0.003076193,0.01299497,0.005438212,0.004238281,0.005437601],"category_scores_gemma":[0.6585836,0.004208228,0.02263562,0.01615468,0.002362705,0.01145772,0.007717223,0.005836986,0.0009850628],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00572399,"about_ca_system_score_gemma":0.01106701,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005212327,"about_ca_topic_score_gemma":0.0147706,"domain_scores_codex":[0.5984443,0.3594617,0.02091673,0.0121719,0.00845223,0.0005532147],"domain_scores_gemma":[0.1089205,0.854475,0.01151018,0.01892998,0.005198289,0.0009660751],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.008604903,0.0007570895,0.05449494,0.06606851,0.1796207,0.0009570966,0.01334069,0.1417491,0.004783763,0.04124238,0.01971381,0.4686671],"study_design_scores_gemma":[0.004110311,0.002149534,0.008422072,0.009504878,0.08720462,0.0006467044,0.002567741,0.6561905,0.007255632,0.1913094,0.02969033,0.0009483921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02689614,0.02249014,0.9174391,0.005374704,0.0006667572,0.006145963,0.008445925,0.01062362,0.001917702],"genre_scores_gemma":[0.1058529,0.001668544,0.8803373,0.0006129828,0.0001509554,0.007281513,0.003186028,0.0006963821,0.0002133973],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6106711,"threshold_uncertainty_score":0.7530662,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4849681266343061,"score_gpt":0.4559516276934493,"score_spread":0.02901649894085673,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}