{"id":"W4386951418","doi":"10.21203/rs.3.rs-3360192/v1","title":"Progression of an Artificial Intelligence Chatbot (ChatGPT) for Pediatric Cardiology Educational Knowledge Assessment: Considerable Gains in a Short Time","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University; SickKids Foundation; Hospital for Sick Children","funders":"","keywords":"Subspecialty; Chatbot; Test (biology); Medicine; Modalities; Stethoscope; Multiple choice; Medical education; Cardiology; Pediatrics; Internal medicine; Artificial intelligence; Computer science; Family medicine; Radiology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005103688,0.0008127533,0.0007017944,0.001017811,0.0003799842,0.001148085,0.001017421,0.001065357,0.004033845],"category_scores_gemma":[0.02849871,0.0002914895,0.0004901703,0.0004365599,0.0003296853,0.001179635,0.001853885,0.0007663099,0.002048037],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005542131,"about_ca_system_score_gemma":0.0009574221,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001950735,"about_ca_topic_score_gemma":0.001851226,"domain_scores_codex":[0.9957181,0.0019037,0.0003425424,0.0007040925,0.0009405791,0.0003910306],"domain_scores_gemma":[0.9801168,0.01106858,0.00154536,0.001729093,0.002772447,0.002767696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005732052,0.01162528,0.07100516,0.0006922125,0.0001449369,0.000504008,0.004022426,0.002863241,0.03332919,0.0002167694,0.007321918,0.8625429],"study_design_scores_gemma":[0.001519413,0.04632897,0.8146973,0.0005154097,0.0004691355,0.001960933,0.00376846,0.0460522,0.0615899,0.0009905676,0.02178434,0.0003233345],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9827865,0.0002140294,0.007825696,0.00036933,0.00008550125,0.0006551893,0.0003224676,0.002240396,0.005500821],"genre_scores_gemma":[0.9711961,0.0001204195,0.02232497,0.0001763048,0.00006513191,0.0006432925,0.0007406691,0.0001217642,0.004611298],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005103688,"threshold_uncertainty_score":0.02699125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.519874502965798,"score_gpt":0.6254870356941005,"score_spread":0.1056125327283025,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}