{"id":"W4234842191","doi":"10.23970/ahrqepcmethmachineperformance","title":"Performance and Usability of Machine Learning for Screening in Systematic Reviews: A Comparative Evaluation of Three Tools","year":2019,"lang":"en","type":"report","venue":"","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Alberta; Agency for Healthcare Research and Quality; U.S. Department of Health and Human Services","keywords":"Usability; Computer science; Usability inspection; Human–computer interaction; Usability engineering; Artificial intelligence; Data science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5060743,0.004584008,0.01338721,0.03103392,0.002696598,0.01087763,0.005051711,0.004654224,0.003227773],"category_scores_gemma":[0.8031161,0.003729149,0.02431985,0.02343522,0.003985846,0.010724,0.00903837,0.003508448,0.0008458473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007354332,"about_ca_system_score_gemma":0.01552993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00301926,"about_ca_topic_score_gemma":0.006537744,"domain_scores_codex":[0.2654355,0.4680088,0.1956974,0.01320464,0.05594137,0.001712373],"domain_scores_gemma":[0.0480373,0.8661824,0.04509939,0.0154713,0.02415548,0.001054026],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.02709421,0.0009854709,0.0551483,0.2646306,0.1001132,0.0003480038,0.007947746,0.006091401,0.001470423,0.002327618,0.009060117,0.524783],"study_design_scores_gemma":[0.04463358,0.02769713,0.1775704,0.2427672,0.3054888,0.00326468,0.005565112,0.1129678,0.01213893,0.0238033,0.03848871,0.005614384],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.2586963,0.3331991,0.2501315,0.01241034,0.003230144,0.1089783,0.01494787,0.009292941,0.009113608],"genre_scores_gemma":[0.3912674,0.02984049,0.5052823,0.00140088,0.0005704091,0.06732947,0.0030238,0.0007258256,0.0005592895],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4939257,"threshold_uncertainty_score":0.6090983,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9298770819181795,"score_gpt":0.6039567952439803,"score_spread":0.3259202866741993,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}