{"id":"W4404963495","doi":"10.2196/64284","title":"Performance Evaluation and Implications of Large Language Models in Radiology Board Exams: Prospective Comparative Analysis","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000775487,0.00007834718,0.0002384305,0.0004134031,0.00003888969,0.000008910113,0.00004681565,0.0001252559,0.0003147157],"category_scores_gemma":[0.0002168155,0.00006780754,0.00004208883,0.0009757307,0.00009121958,0.0001504758,0.00001191069,0.0002101293,0.00001031366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002205792,"about_ca_system_score_gemma":0.001580728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003873585,"about_ca_topic_score_gemma":0.0003690163,"domain_scores_codex":[0.9988438,0.00009753414,0.0003810793,0.0002538475,0.0002809012,0.0001428747],"domain_scores_gemma":[0.9992763,0.0001413409,0.00006039396,0.0001669121,0.0002269848,0.0001280054],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001095743,0.001069913,0.4163324,0.0004301861,0.0002124846,0.000001048714,0.1025893,0.0001312057,0.0003854756,0.009214686,0.001352619,0.4681711],"study_design_scores_gemma":[0.0001049661,0.0001679009,0.7187861,0.000289051,0.0002880763,0.00001607337,0.01334675,0.2636933,0.0006799035,0.002334638,0.0002008422,0.00009242818],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9898356,0.002617925,0.0004123509,0.003955245,0.0001962594,0.0008245445,0.000003944054,0.00002489253,0.002129214],"genre_scores_gemma":[0.9982944,0.0002952551,0.0001413636,0.0002684005,0.0001660165,0.0005918182,0.0001516272,0.000005066943,0.00008602387],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4680786,"threshold_uncertainty_score":0.3445916,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09867950018835837,"score_gpt":0.5082866943012985,"score_spread":0.4096071941129401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}