{"id":"W4400983681","doi":"10.1002/jrsm.1736","title":"Considerations for conducting systematic reviews: A follow‐up study to evaluate the performance of various automated methods for reference de‐duplication","year":2024,"lang":"en","type":"article","venue":"Research Synthesis Methods","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University; Queen's University","funders":"","keywords":"Computer science; Evaluation methods; Research methodology; Reliability engineering; Engineering; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6645209,0.001869497,0.003579222,0.01066236,0.006106209,0.01119917,0.005354474,0.00977711,0.008823236],"category_scores_gemma":[0.9038054,0.002943151,0.01316457,0.01433062,0.003811145,0.02535789,0.007595101,0.008999639,0.002285825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02124646,"about_ca_system_score_gemma":0.05956944,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008977263,"about_ca_topic_score_gemma":0.02529553,"domain_scores_codex":[0.2982846,0.4704328,0.1297298,0.009130215,0.08707704,0.005345376],"domain_scores_gemma":[0.04329069,0.7438792,0.03440589,0.04042581,0.1355009,0.002497531],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.009393701,0.001610745,0.03643531,0.1852096,0.0107355,0.001852817,0.04289082,0.002344508,0.006587559,0.01978328,0.09490868,0.5882475],"study_design_scores_gemma":[0.008714831,0.01357251,0.05160249,0.2715829,0.04188906,0.004008599,0.02687616,0.01304649,0.02620563,0.04590558,0.4942968,0.002298894],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.1553989,0.07998042,0.2429647,0.2648188,0.01981856,0.1967113,0.01269588,0.003594005,0.02401749],"genre_scores_gemma":[0.2623783,0.01403453,0.4747364,0.08093832,0.00221555,0.1591033,0.002681976,0.001030279,0.002881386],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3354791,"threshold_uncertainty_score":0.4137055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9579403401628281,"score_gpt":0.7454122149889704,"score_spread":0.2125281251738578,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}