{"id":"W4396646848","doi":"10.2196/59267","title":"Evaluating ChatGPT-4’s Accuracy in Identifying Final Diagnoses Within Differential Diagnoses Compared With Those of Physicians: Experimental Study for Diagnostic Cases","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Dokkyo Medical University","keywords":"Medical diagnosis; Differential (mechanical device); Diagnostic accuracy; Medicine; Medical physics; Radiology; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00128537,0.0002064761,0.0004415654,0.0005839829,0.0002803997,0.0001401565,0.0001711322,0.0000643077,0.00007930075],"category_scores_gemma":[0.003061075,0.0001598988,0.00008735366,0.0008342027,0.0002495386,0.0006285842,0.0000963873,0.000511341,0.00002314873],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000317184,"about_ca_system_score_gemma":0.0004900779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001421669,"about_ca_topic_score_gemma":0.0005720721,"domain_scores_codex":[0.9969203,0.0004115789,0.000757685,0.0003967131,0.0009760071,0.0005376887],"domain_scores_gemma":[0.9867748,0.01217112,0.0001392344,0.0002699513,0.000549677,0.00009521714],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.004245947,0.01631388,0.09281886,0.005230396,0.0003489311,0.000377145,0.7415237,0.00025929,0.00824385,0.000797944,0.001622489,0.1282175],"study_design_scores_gemma":[0.001508908,0.02350092,0.1246818,0.01232776,0.0001218542,0.00005857899,0.5859408,0.02305628,0.2272922,0.0009374873,0.00008121474,0.0004921851],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9930699,0.001444127,0.0001758261,0.0003583121,0.0003696389,0.004459404,0.00003385364,0.00004508312,0.00004389442],"genre_scores_gemma":[0.9947692,0.00009377103,0.0001573953,0.00002311602,0.0002924399,0.004526145,0.00006536481,0.00003453975,0.00003807154],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2190483,"threshold_uncertainty_score":0.6520483,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5837778148836222,"score_gpt":0.6306624332768651,"score_spread":0.04688461839324287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}