{"id":"W4403097762","doi":"10.1001/jamaoncol.2024.4327","title":"Ensuring Safety and Consistency in Artificial Intelligence Chatbot Responses—Reply","year":2024,"lang":"en","type":"article","venue":"JAMA Oncology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto","funders":"","keywords":"Chatbot; Medicine; Consistency (knowledge bases); Natural language processing; Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009941876,0.0007964012,0.001080229,0.001084895,0.00519646,0.004642201,0.002691681,0.04999642,0.006884987],"category_scores_gemma":[0.09008393,0.001117614,0.001255996,0.000989674,0.005096019,0.005388944,0.003066173,0.0582448,0.005436704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005858891,"about_ca_system_score_gemma":0.009724393,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01928191,"about_ca_topic_score_gemma":0.02401396,"domain_scores_codex":[0.9926816,0.001608543,0.0012403,0.001716193,0.001832184,0.0009212389],"domain_scores_gemma":[0.9377299,0.04003584,0.001914218,0.00142793,0.01505322,0.003838949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001100669,0.00005167578,0.001429299,0.0001100802,0.00005495353,0.0004662747,0.0009278234,0.0001183535,0.0004602071,0.002729969,0.9858966,0.007644724],"study_design_scores_gemma":[0.0003288653,0.0001833467,0.007975443,0.0008833207,0.0002455577,0.001394709,0.006247964,0.00111552,0.001868972,0.01464699,0.9646704,0.0004389953],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0003631334,0.0003434283,0.0002347189,0.9904603,0.008035388,0.00001337198,0.00006300126,0.00003715876,0.00044947],"genre_scores_gemma":[0.002917942,0.0003240048,0.0003017785,0.9893509,0.005523272,0.00005147139,0.00002645213,0.00001591132,0.001488303],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.04999642,"threshold_uncertainty_score":0.05257827,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1866560307679856,"score_gpt":0.4587237640772963,"score_spread":0.2720677333093107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}