{"id":"W4411798619","doi":"10.1101/2025.06.28.25330485","title":"Evaluation of Closed and Open Large Language Models in Pediatric Cardiology Board Exam Performance","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Health Promotion and Cardiovascular Prevention","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children","funders":"","keywords":"Cardiology; Internal medicine; Medicine; Medical physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01179938,0.001066815,0.0006519346,0.001131984,0.0002974266,0.00270573,0.00137054,0.0008938173,0.003741823],"category_scores_gemma":[0.08974227,0.0003374102,0.001110477,0.0006288328,0.000518578,0.002256575,0.002641039,0.00139963,0.001401919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001370315,"about_ca_system_score_gemma":0.001645149,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0039101,"about_ca_topic_score_gemma":0.004128265,"domain_scores_codex":[0.9913017,0.004943029,0.000680693,0.001411388,0.001394898,0.0002683678],"domain_scores_gemma":[0.8971437,0.0848181,0.005685534,0.003428458,0.00620207,0.00272212],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.007748778,0.003867829,0.446155,0.001711089,0.0008033874,0.0004598092,0.005749571,0.06010269,0.006666954,0.001402456,0.01609648,0.449236],"study_design_scores_gemma":[0.001111033,0.008083434,0.3584075,0.001274422,0.001007619,0.000860322,0.003834084,0.5854443,0.02030253,0.004598492,0.01464062,0.0004356093],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9799397,0.0004072884,0.01158812,0.0003841484,0.00009026781,0.0004345315,0.00163857,0.001841651,0.003675713],"genre_scores_gemma":[0.9785598,0.0001693753,0.01664118,0.0001357871,0.00003868317,0.0004102943,0.002780021,0.0001822002,0.001082669],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01179938,"threshold_uncertainty_score":0.06240183,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06584730613277463,"score_gpt":0.3783727674758237,"score_spread":0.3125254613430491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}