{"id":"W4385336911","doi":"10.1016/j.jacr.2023.07.010","title":"Using Artificial Intelligence Chatbots as a Radiologic Decision-Making Tool for Liver Imaging: Do ChatGPT and Bard Communicate Information Consistent With the ACR Appropriateness Criteria?","year":2023,"lang":"en","type":"article","venue":"Journal of the American College of Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":24,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University Medical Centre; Juravinski Hospital; University of Toronto; McMaster University","funders":"","keywords":"Workflow; Clinical decision making; Medical decision making; Computer science; Medicine; Artificial intelligence; Medical emergency; Family medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01976299,0.0007607747,0.0006410381,0.001464308,0.001632289,0.003304135,0.001997159,0.003195264,0.0120718],"category_scores_gemma":[0.1179554,0.0004791755,0.0004200886,0.0005919709,0.001292711,0.003958495,0.002590818,0.002305384,0.003843754],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001086834,"about_ca_system_score_gemma":0.001641755,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001403898,"about_ca_topic_score_gemma":0.001509268,"domain_scores_codex":[0.9838424,0.01353561,0.0005251225,0.00066703,0.0009260831,0.0005038999],"domain_scores_gemma":[0.7816322,0.1972421,0.00744193,0.003853982,0.005284952,0.004544837],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01097318,0.006404283,0.1071543,0.002270611,0.0004997195,0.002926186,0.01963017,0.009157694,0.01460285,0.01711276,0.07083093,0.7384373],"study_design_scores_gemma":[0.003948433,0.01036073,0.1258328,0.004971999,0.001606132,0.006186891,0.03322271,0.4350255,0.05212935,0.1477309,0.1773891,0.001595555],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6810703,0.002633443,0.1966159,0.03688571,0.002170749,0.001240598,0.001168299,0.01532627,0.06288871],"genre_scores_gemma":[0.9398502,0.0003227511,0.0509493,0.003687868,0.0003189414,0.0004427793,0.000343741,0.0002454575,0.003838892],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.980237,"threshold_uncertainty_score":0.1045179,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1126033534804542,"score_gpt":0.4094638704986628,"score_spread":0.2968605170182086,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}