{"id":"W4410449531","doi":"10.2196/72034","title":"Assessment of Large Language Model Performance on Medical School Essay-Style Concept Appraisal Questions: Exploratory Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Style (visual arts); Exploratory research; Psychology; Computer science; Sociology; Social science; Art; Visual arts; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01703307,0.0007528092,0.0007325449,0.001243504,0.0006889054,0.001890641,0.0008850633,0.0008970444,0.003234576],"category_scores_gemma":[0.08997589,0.0002741667,0.0004864538,0.0006161383,0.0008953929,0.001543824,0.001955151,0.001242126,0.002063979],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000674416,"about_ca_system_score_gemma":0.0009930928,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005642841,"about_ca_topic_score_gemma":0.0009453614,"domain_scores_codex":[0.9896248,0.00653299,0.0009550728,0.001053883,0.001402605,0.0004307023],"domain_scores_gemma":[0.8838977,0.08325049,0.01036633,0.004474153,0.01377612,0.00423528],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006738828,0.0162561,0.5700028,0.001137089,0.0002664472,0.00106644,0.08973241,0.003018013,0.02809341,0.0007338725,0.00546462,0.27749],"study_design_scores_gemma":[0.0009549177,0.03484658,0.8470392,0.0004516325,0.0002563282,0.002173044,0.03720958,0.02560855,0.03826397,0.002577078,0.01031556,0.0003034735],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9962847,0.00006204836,0.00174964,0.00008252174,0.00001466666,0.0003226254,0.000100839,0.00006641453,0.001316574],"genre_scores_gemma":[0.9942235,0.00006381126,0.003837429,0.00009116792,0.00001905041,0.0004244955,0.0002190557,0.00002814023,0.001093243],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9829669,"threshold_uncertainty_score":0.09008056,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04890214882415592,"score_gpt":0.5064443918100545,"score_spread":0.4575422429858986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}