{"id":"W4311481338","doi":"10.1016/j.jval.2022.09.1805","title":"MSR74 Can Artificial Intelligence (AI) Replace a Human Reviewer in Systematic Literature Review (SLR)? Validation of the LIVESTARTTM Tool","year":2022,"lang":"en","type":"article","venue":"Value in Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Inclusion and exclusion criteria; Inclusion (mineral); Computer science; Artificial intelligence; Systematic review; Machine learning; Medical physics; Population; Medicine; Natural language processing; Information retrieval; MEDLINE; Alternative medicine; Psychology; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6741436,0.001642644,0.006577126,0.01391361,0.004594714,0.01084946,0.007078663,0.008168166,0.02226875],"category_scores_gemma":[0.892987,0.003239492,0.009095896,0.01222443,0.01075085,0.01131421,0.01056967,0.004541888,0.004055778],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008009111,"about_ca_system_score_gemma":0.05235551,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005835477,"about_ca_topic_score_gemma":0.0207513,"domain_scores_codex":[0.2373506,0.5240303,0.1633931,0.02549977,0.04571572,0.004010531],"domain_scores_gemma":[0.05827732,0.7550566,0.05727433,0.05301701,0.07299554,0.003379238],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01359991,0.0008313909,0.06775897,0.3095663,0.03021648,0.001147838,0.0488131,0.001807488,0.003051034,0.04537971,0.1725214,0.3053065],"study_design_scores_gemma":[0.03175498,0.003944195,0.06559188,0.2235083,0.04199994,0.001725451,0.01588993,0.01557265,0.007096631,0.06821383,0.5229588,0.001743441],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1258148,0.05232421,0.2790973,0.2310631,0.03448709,0.1713473,0.03945166,0.005799127,0.0606154],"genre_scores_gemma":[0.3640049,0.002944764,0.3239658,0.04498832,0.002310469,0.2499987,0.005900326,0.00118524,0.004701619],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3258564,"threshold_uncertainty_score":0.4018391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1691173899103994,"score_gpt":0.4281731953302321,"score_spread":0.2590558054198326,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}