{"id":"W7136785886","doi":"10.4103/apc.apc_301_25","title":"Comparing closed and open large language models on pediatric cardiology board exam performance","year":2025,"lang":"en","type":"article","venue":"Annals of Pediatric Cardiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children","funders":"","keywords":"Subspecialty; Clinical cardiology; Editorial board; Pediatric Radiology; MEDLINE; Clinical Practice","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007372592,0.001438409,0.0006457394,0.00112834,0.0002715727,0.001927044,0.001140208,0.0009804351,0.002533484],"category_scores_gemma":[0.0446421,0.0003506127,0.0008337219,0.0006566497,0.0004823511,0.003001699,0.002113235,0.001875715,0.001638345],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001266919,"about_ca_system_score_gemma":0.001078634,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007502801,"about_ca_topic_score_gemma":0.007553144,"domain_scores_codex":[0.9954414,0.002538192,0.0003459206,0.0008820791,0.0006010755,0.0001912705],"domain_scores_gemma":[0.9480467,0.04510725,0.001342484,0.001739035,0.00263842,0.001126059],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006988233,0.002764659,0.1696175,0.001038791,0.001003469,0.0004474798,0.002185893,0.1819842,0.00681962,0.001823608,0.02118306,0.6041434],"study_design_scores_gemma":[0.0004467755,0.003200481,0.07092387,0.0003582645,0.0004966348,0.0003983653,0.001190208,0.9034041,0.008198586,0.004316953,0.006855669,0.0002101623],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9548408,0.001926227,0.02955521,0.0008603795,0.0002073235,0.0002627529,0.00239753,0.003713251,0.00623655],"genre_scores_gemma":[0.9709697,0.0004728733,0.02050233,0.0002603785,0.00008396873,0.0002039307,0.005366268,0.000255088,0.00188544],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007502801,"threshold_uncertainty_score":0.03899044,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2416587721565181,"score_gpt":0.4488845448533944,"score_spread":0.2072257726968763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}