{"id":"W4402171959","doi":"10.2196/51319","title":"Assessing the Current Limitations of Large Language Models in Advancing Health Care Education","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Health care; Engineering ethics; Computer science; Political science; Engineering; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1642328,0.001259274,0.00106269,0.003312493,0.001729924,0.01883947,0.005185392,0.003523964,0.008347264],"category_scores_gemma":[0.4975706,0.001179817,0.001451754,0.002804153,0.007060923,0.0243411,0.01117019,0.005827939,0.002447092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005286772,"about_ca_system_score_gemma":0.01164841,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007397753,"about_ca_topic_score_gemma":0.008044823,"domain_scores_codex":[0.8498055,0.1257654,0.004998796,0.004061253,0.01401636,0.00135274],"domain_scores_gemma":[0.3156585,0.6242229,0.01036515,0.02919316,0.0179983,0.00256202],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001508247,0.0007765965,0.04700031,0.004631506,0.0005721967,0.0002429191,0.01484142,0.05098962,0.00183564,0.3082639,0.01941271,0.5499251],"study_design_scores_gemma":[0.0002694837,0.0008966784,0.007170839,0.009281985,0.0004759134,0.0004581646,0.01144584,0.2632766,0.006213278,0.5693434,0.1308565,0.000311118],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09850693,0.01561706,0.6461459,0.1872493,0.001232496,0.001074877,0.002046365,0.003172204,0.04495487],"genre_scores_gemma":[0.6207332,0.006184611,0.3600951,0.007123663,0.0007659221,0.001231053,0.001058671,0.0006520652,0.002155669],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1642328,"threshold_uncertainty_score":0.8685565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3951358902688075,"score_gpt":0.6376676475787504,"score_spread":0.2425317573099429,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}