{"id":"W4399700451","doi":"10.1016/j.jval.2024.03.171","title":"CO83 Screening Articles in a Qualitative Literature Review Using Large Language Models: A Comparison of GPT Versus Open Source, Trained Models Against Expert Researcher Screening","year":2024,"lang":"en","type":"article","venue":"Value in Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Open source; Medical physics; Data science; Natural language processing; Psychology; Medicine; Programming language; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1229803,0.0009061978,0.002559505,0.01636807,0.00208928,0.004131082,0.00210719,0.001772847,0.01286429],"category_scores_gemma":[0.4817515,0.0009234609,0.003459283,0.01290892,0.001659652,0.005659817,0.004701232,0.001082392,0.001849809],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006800684,"about_ca_system_score_gemma":0.02766145,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008955179,"about_ca_topic_score_gemma":0.02366969,"domain_scores_codex":[0.904259,0.05693066,0.01988933,0.004014739,0.0131946,0.001711752],"domain_scores_gemma":[0.3127115,0.6030475,0.03032212,0.007892654,0.04449521,0.001530923],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003613345,0.0004899934,0.03622805,0.3187538,0.003509803,0.001805917,0.09373163,0.001228326,0.003876215,0.00639302,0.02759347,0.5027764],"study_design_scores_gemma":[0.00459922,0.00321101,0.0992263,0.3876396,0.03606278,0.00275224,0.1659821,0.008608541,0.0150769,0.02043904,0.255427,0.0009752504],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5312684,0.09588426,0.1000295,0.03227467,0.001680592,0.1076376,0.0555243,0.001637866,0.0740627],"genre_scores_gemma":[0.7246682,0.02480496,0.1370699,0.007280975,0.0003197518,0.08869227,0.007816838,0.0004907534,0.008856418],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8770197,"threshold_uncertainty_score":0.6503898,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6808479755722152,"score_gpt":0.6032836486369278,"score_spread":0.07756432693528736,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}