{"id":"W4288257341","doi":"10.48550/arxiv.1908.08610","title":"Viability of machine learning to reduce workload in systematic review\\n screenings in the health sciences: a working paper","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"","keywords":"Machine learning; Workload; Artificial intelligence; Computer science; Support vector machine; Systematic review; Classifier (UML); Naive Bayes classifier; Inclusion and exclusion criteria; Automation; MEDLINE; Medicine; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.414103,0.003569271,0.007654272,0.006628479,0.002482773,0.008345914,0.004242731,0.004226009,0.004200658],"category_scores_gemma":[0.6920634,0.003643073,0.009116371,0.00960756,0.003395947,0.01892951,0.005343109,0.00533021,0.001754697],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006956262,"about_ca_system_score_gemma":0.02274079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007470558,"about_ca_topic_score_gemma":0.01110826,"domain_scores_codex":[0.4942565,0.4458395,0.03011268,0.008374687,0.01980145,0.001615126],"domain_scores_gemma":[0.1174021,0.8062033,0.01730916,0.03519002,0.02204311,0.001852258],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.009425651,0.0007442681,0.0105589,0.0260649,0.00653577,0.0001937666,0.002579443,0.03940814,0.002049683,0.006785844,0.02244219,0.8732114],"study_design_scores_gemma":[0.008947354,0.007412992,0.01846361,0.02143487,0.009769782,0.0009510508,0.002033535,0.7172223,0.009551943,0.152341,0.05089501,0.00097652],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1315084,0.0956491,0.6028469,0.1194941,0.005950807,0.01750692,0.005436506,0.01158875,0.01001842],"genre_scores_gemma":[0.3023861,0.008250246,0.6688554,0.006534485,0.00158976,0.008168804,0.002427144,0.0006927886,0.001095275],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.585897,"threshold_uncertainty_score":0.7225153,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6372670749371275,"score_gpt":0.3906863311464441,"score_spread":0.2465807437906834,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}