{"id":"W2982683456","doi":"10.1186/s13643-019-1222-2","title":"Performance and usability of machine learning for screening in systematic reviews: a comparative evaluation of three tools","year":2019,"lang":"en","type":"article","venue":"Systematic Reviews","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":132,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Agency for Healthcare Research and Quality; U.S. Department of Health and Human Services","keywords":"Medicine; Usability; Medical physics; MEDLINE; Medical education; Artificial intelligence; Human–computer interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3501187,0.002579106,0.005689669,0.01661103,0.002222188,0.006423134,0.004726471,0.002438123,0.001738768],"category_scores_gemma":[0.6321213,0.002812871,0.008440601,0.01480132,0.002948099,0.008136827,0.007481618,0.0019385,0.0004120368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00877637,"about_ca_system_score_gemma":0.01622207,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004284056,"about_ca_topic_score_gemma":0.0105268,"domain_scores_codex":[0.4809763,0.3719957,0.1005281,0.01042191,0.03420017,0.001877855],"domain_scores_gemma":[0.07709578,0.8507479,0.04027208,0.01362272,0.01659516,0.001666274],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0288062,0.003042291,0.0845296,0.100256,0.01545699,0.000448775,0.0245756,0.0112776,0.002174996,0.001137554,0.004479807,0.7238147],"study_design_scores_gemma":[0.04732014,0.08651322,0.380529,0.09202977,0.05242913,0.005185978,0.02025091,0.2508528,0.01832237,0.01149728,0.02933182,0.005737601],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7845381,0.04506221,0.09978816,0.00566166,0.0006095528,0.0455227,0.00422354,0.008418558,0.006175521],"genre_scores_gemma":[0.5230355,0.007230093,0.441736,0.0006228181,0.0001495286,0.02554547,0.001111662,0.0002653922,0.0003035927],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6498812,"threshold_uncertainty_score":0.8014193,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8283611526367183,"score_gpt":0.548349083404606,"score_spread":0.2800120692321123,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}