{"id":"W4391225179","doi":"10.1093/jamia/ocad252","title":"Harnessing the potential of large language models in medical education: promise and pitfalls","year":2024,"lang":"en","type":"review","venue":"Journal of the American Medical Informatics Association","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":89,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University Health Centre","funders":"National Key Research and Development Program of China; National Institutes of Health; National Natural Science Foundation of China","keywords":"Narrative; Misconduct; Process (computing); Engineering ethics; Psychology; Medicine; Medical education; Political science; Computer science; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0138609,0.0006529447,0.00136409,0.002675351,0.0004823821,0.004972745,0.001567032,0.002263742,0.003261992],"category_scores_gemma":[0.02302627,0.0005565624,0.001330309,0.001989973,0.002391729,0.008070602,0.003270202,0.004088576,0.001284512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001233351,"about_ca_system_score_gemma":0.006229221,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002676328,"about_ca_topic_score_gemma":0.004736848,"domain_scores_codex":[0.9945851,0.003270393,0.0004396797,0.000273446,0.001263361,0.0001680961],"domain_scores_gemma":[0.9608026,0.03509348,0.001146553,0.0008734455,0.001718351,0.0003655456],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005211647,0.00008600522,0.0004098605,0.01713264,0.0001676079,0.0001098798,0.0005707985,0.0006460223,0.0004455503,0.01819465,0.005462835,0.9567221],"study_design_scores_gemma":[0.00007664071,0.0004430524,0.001659431,0.06358122,0.0005175314,0.001853093,0.001368369,0.001190683,0.001707687,0.03586619,0.8916225,0.0001137015],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0007082891,0.9845483,0.003489879,0.006619507,0.0002822221,0.00003798904,0.00002071616,0.00004472708,0.004248333],"genre_scores_gemma":[0.013979,0.9735651,0.008343503,0.00283935,0.0004911493,0.0001274038,0.00003354925,0.00002648412,0.0005944683],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.0138609,"threshold_uncertainty_score":0.0733043,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05065315228572722,"score_gpt":0.4462854854152415,"score_spread":0.3956323331295143,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}