{"id":"W4388801023","doi":"10.1101/2023.11.15.23298499","title":"Spot the Difference: Can ChatGPT4-Vision Transform Radiology Artificial Intelligence?","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Science Foundation Ireland; Health Service Executive; Royal College of Surgeons in Ireland; Wellcome Trust; Canadian Institute for Theoretical Astrophysics","keywords":"Computer science; Artificial intelligence; Coding (social sciences); Recall; Machine learning; F1 score; Task (project management); Psychology; Statistics; Cognitive psychology; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003898832,0.0009970855,0.0004674967,0.0007931942,0.0003754747,0.001875798,0.002564498,0.001868732,0.01826032],"category_scores_gemma":[0.01650259,0.000487491,0.0006604458,0.0004464524,0.0008478626,0.003845962,0.002380459,0.001680054,0.008193376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009521773,"about_ca_system_score_gemma":0.0008176694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006242537,"about_ca_topic_score_gemma":0.005676638,"domain_scores_codex":[0.9986042,0.0006530254,0.00007130585,0.0002500946,0.0002905637,0.0001308019],"domain_scores_gemma":[0.9942621,0.003231783,0.0002080606,0.001054228,0.0009470748,0.0002968833],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002373286,0.0004378629,0.00852639,0.0007104624,0.0001943519,0.0009589478,0.001095347,0.03976234,0.01859336,0.01484958,0.1675268,0.7449713],"study_design_scores_gemma":[0.0002297044,0.0007067058,0.003998409,0.0002623579,0.00008388358,0.0007106385,0.0005894547,0.7965134,0.0476599,0.05774424,0.091327,0.0001742691],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.1021177,0.00155098,0.5922013,0.01024513,0.001898321,0.0004166479,0.003202736,0.2586889,0.02967842],"genre_scores_gemma":[0.6603957,0.000548113,0.301473,0.002774415,0.0003332362,0.0003788334,0.004276202,0.005565577,0.02425503],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.01826032,"threshold_uncertainty_score":0.06108689,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09104601245486954,"score_gpt":0.3611499249292693,"score_spread":0.2701039124743997,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}