{"id":"W4378190436","doi":"10.1002/jrsm.1636","title":"A real‐world evaluation of the implementation of <scp>NLP</scp> technology in abstract screening of a systematic review","year":2023,"lang":"en","type":"review","venue":"Research Synthesis Methods","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; McGill University; The Quebec Population Health Research Network; University of Calgary; McGill University Health Centre","funders":"Canadian Medical Association; Robert Koch Institut; Koch Institute for Integrative Cancer Research, Massachusetts Institute of Technology; Public Health Agency; Rhodes Scholarships; Public Health Agency of Canada; World Health Organization","keywords":"Computer science; Context (archaeology); Systematic review; Recall; Inclusion (mineral); Machine learning; Artificial intelligence; Precision and recall; Natural language processing; MEDLINE; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6654018,0.002730899,0.004437929,0.01095108,0.002604285,0.00969737,0.004488717,0.005467197,0.005412652],"category_scores_gemma":[0.8245941,0.003386607,0.01013229,0.01297528,0.003717944,0.008328497,0.006435967,0.003286818,0.001309114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01141324,"about_ca_system_score_gemma":0.02242014,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003718913,"about_ca_topic_score_gemma":0.007621762,"domain_scores_codex":[0.2026139,0.6693611,0.09146886,0.00798778,0.02689668,0.00167174],"domain_scores_gemma":[0.05942923,0.802035,0.05098222,0.0403363,0.04565741,0.00155978],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.03744539,0.002706229,0.02507137,0.2865123,0.02619231,0.0009784647,0.01857415,0.01534871,0.01136487,0.00747868,0.02568528,0.5426422],"study_design_scores_gemma":[0.111104,0.09702618,0.1306114,0.2060198,0.09417476,0.00413133,0.008991586,0.09661832,0.03798469,0.02140876,0.1868075,0.005121847],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3133653,0.05279097,0.3214125,0.02795133,0.003631315,0.2315404,0.01883638,0.01045005,0.02002171],"genre_scores_gemma":[0.2857217,0.005595898,0.5881714,0.002969988,0.0003538497,0.1137647,0.002274669,0.0006137548,0.0005340753],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3345982,"threshold_uncertainty_score":0.4126192,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9514931780577904,"score_gpt":0.7593710306945818,"score_spread":0.1921221473632087,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}