{"id":"W4411798619","doi":"10.1101/2025.06.28.25330485","title":"Evaluation of Closed and Open Large Language Models in Pediatric Cardiology Board Exam Performance","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Health Promotion and Cardiovascular Prevention","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children","funders":"","keywords":"Cardiology; Internal medicine; Medicine; Medical physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01613536,0.0001487911,0.0007251695,0.0003001033,0.00002819404,0.000009402086,0.0001702871,0.0003185088,0.00003874787],"category_scores_gemma":[0.0002445196,0.0001429253,0.0001462999,0.0001906001,0.00001936674,0.00006853551,0.0005095141,0.0004692,0.000003949524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001260925,"about_ca_system_score_gemma":0.0007254677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009767431,"about_ca_topic_score_gemma":0.00004120941,"domain_scores_codex":[0.9963441,0.002035478,0.0004727939,0.0004048061,0.000540369,0.0002024225],"domain_scores_gemma":[0.9989185,0.00003085296,0.0001523004,0.0005281102,0.000299727,0.00007044651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001259686,0.0003884854,0.7166417,0.01174158,0.0006256156,0.00004516157,0.003501971,0.006890207,0.0001379201,0.0002170567,0.0002184329,0.2594659],"study_design_scores_gemma":[0.003979159,0.00008280008,0.9533252,0.0004670092,0.001053099,0.000009708563,0.0001105271,0.04010177,0.00009789752,0.0005043346,0.0001310413,0.0001374722],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9806065,0.00479317,0.002322962,0.0001310488,0.0003010251,0.00209343,0.00002045966,0.00002061301,0.009710804],"genre_scores_gemma":[0.9956311,0.003285104,0.0002428102,0.0000664676,0.0001411098,0.0002451703,0.00009493652,0.00001118988,0.0002821005],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2593284,"threshold_uncertainty_score":0.5828323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06584730613277463,"score_gpt":0.3783727674758237,"score_spread":0.3125254613430491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}