{"id":"W4403321301","doi":"10.1186/s12909-024-06115-5","title":"The future of AI clinicians: assessing the modern standard of chatbots and their approach to diagnostic uncertainty","year":2024,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"","keywords":"Medicine; Correctness; MEDLINE; Interpretation (philosophy); Family medicine; Medical physics; Medical education; Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001516389,0.00008852621,0.0001868755,0.00005602534,0.0001253151,0.00003903898,0.000117914,0.0001158413,0.00002391119],"category_scores_gemma":[0.004896178,0.00004302185,0.0000547467,0.000279395,0.000196703,0.00006199383,0.00002347652,0.0002958493,0.000002773972],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006157758,"about_ca_system_score_gemma":0.004854208,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002281486,"about_ca_topic_score_gemma":0.000198979,"domain_scores_codex":[0.9986721,0.0001472871,0.0004706655,0.0001875763,0.0003750937,0.0001472555],"domain_scores_gemma":[0.9961374,0.003061415,0.00007866361,0.0002724733,0.0002715655,0.0001784206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.00005544309,0.0001639313,0.005087901,0.0006350276,0.0000194314,1.317689e-7,0.01106991,0.00002911841,0.00002464261,0.006194175,0.00590837,0.9708119],"study_design_scores_gemma":[0.0003509082,0.001977345,0.07002718,0.01282563,0.0005753998,0.0001937124,0.3662086,0.2047318,0.005243477,0.1861281,0.1510286,0.0007091779],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8298858,0.02534107,0.03662412,0.1017979,0.00415358,0.001144789,0.000008286802,0.0000386831,0.0010058],"genre_scores_gemma":[0.9951826,0.001261026,0.0004132223,0.001553859,0.001383788,0.00008675703,0.00001828008,0.00001123994,0.00008920964],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9701027,"threshold_uncertainty_score":0.8611156,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08794833893524416,"score_gpt":0.4630393188298464,"score_spread":0.3750909798946022,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}