{"id":"W4400589155","doi":"10.1007/s00330-024-10902-5","title":"ChatGPT’s diagnostic performance based on textual vs. visual information compared to radiologists’ diagnostic performance in musculoskeletal radiology","year":2024,"lang":"en","type":"article","venue":"European Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Guerbet","keywords":"Medicine; Neuroradiology; Interventional radiology; Diagnostic accuracy; Radiology; Medical diagnosis; Neurology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008133836,0.0009019438,0.0006852578,0.00242223,0.0003467235,0.001131219,0.001476276,0.001716757,0.002594817],"category_scores_gemma":[0.04736794,0.0003615492,0.001069432,0.0007097364,0.0006986298,0.001728283,0.00213943,0.0008052798,0.001128832],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007318451,"about_ca_system_score_gemma":0.0006056432,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001947625,"about_ca_topic_score_gemma":0.003022734,"domain_scores_codex":[0.9957925,0.001873866,0.0004386319,0.001053518,0.0006186601,0.000222729],"domain_scores_gemma":[0.9717106,0.02070455,0.001694539,0.002046432,0.002801113,0.001042745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00825538,0.0005577579,0.6294057,0.001123579,0.0009092196,0.001992674,0.001245808,0.02136731,0.01555851,0.0005733725,0.007780261,0.3112304],"study_design_scores_gemma":[0.0008406532,0.005213279,0.5610716,0.0005057082,0.001744031,0.0185246,0.001819494,0.341443,0.05628847,0.005029159,0.007144461,0.0003756519],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9702259,0.001410703,0.02143602,0.0003595572,0.0002103639,0.0003030003,0.001542191,0.001758697,0.00275358],"genre_scores_gemma":[0.9842398,0.0002027377,0.01269795,0.0001268546,0.00008913931,0.00009250035,0.001865239,0.00007584436,0.0006098922],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008133836,"threshold_uncertainty_score":0.04301637,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04325767552305441,"score_gpt":0.3517228958222935,"score_spread":0.3084652202992391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}