{"id":"W4404963495","doi":"10.2196/64284","title":"Performance Evaluation and Implications of Large Language Models in Radiology Board Exams: Prospective Comparative Analysis","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02713548,0.001207424,0.0008620642,0.002124671,0.0003896343,0.002166234,0.001136578,0.00107825,0.00129945],"category_scores_gemma":[0.07883425,0.0004064351,0.001808639,0.001126181,0.0006880476,0.002410623,0.001462968,0.001364211,0.0008143465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001905228,"about_ca_system_score_gemma":0.001361786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008481173,"about_ca_topic_score_gemma":0.00639515,"domain_scores_codex":[0.9882421,0.008538154,0.0007104317,0.001197759,0.001059851,0.000251864],"domain_scores_gemma":[0.8831797,0.1025177,0.003782974,0.003839719,0.00495059,0.00172934],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006577095,0.001826196,0.7534859,0.0005796406,0.001920556,0.0004055367,0.001785674,0.07931931,0.001913499,0.000767156,0.003623158,0.1477963],"study_design_scores_gemma":[0.0005317866,0.006714708,0.304384,0.000329602,0.001343576,0.001095075,0.002488005,0.6715401,0.00494742,0.002501959,0.003906098,0.0002177276],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9885772,0.0011332,0.007089211,0.0004694111,0.00004993637,0.0001586418,0.001060874,0.000313618,0.001147848],"genre_scores_gemma":[0.9932821,0.0002924127,0.004210797,0.00007428086,0.000029866,0.00007880906,0.001692093,0.00004286597,0.0002968505],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02713548,"threshold_uncertainty_score":0.1435078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09867950018835837,"score_gpt":0.5082866943012985,"score_spread":0.4096071941129401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}