{"id":"W7161558043","doi":"10.1093/postmj/qgaf195","title":"Artificial intelligence for assessment in competency-based medical education: current practices and future directions","year":2025,"lang":"en","type":"article","venue":"Postgraduate Medical Journal","topic":"Innovations in Medical Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University; University of British Columbia","funders":"","keywords":"Software deployment; Scopus; Narrative review; Inclusion (mineral); MEDLINE; Educational measurement; Applications of artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002868378,0.0001757441,0.0003341006,0.0006219012,0.0002911611,0.00007831917,0.000222778,0.0002271453,0.0006019933],"category_scores_gemma":[0.01219078,0.0001417353,0.0000857347,0.0009749007,0.0002827133,0.0001273081,0.00004190746,0.001893288,0.000004870914],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003090074,"about_ca_system_score_gemma":0.02596053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000234041,"about_ca_topic_score_gemma":0.00009283584,"domain_scores_codex":[0.9968957,0.0001748283,0.001006698,0.0003088881,0.001291421,0.0003224245],"domain_scores_gemma":[0.9978955,0.0004455262,0.0003682483,0.0001969039,0.0006774869,0.0004162931],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001011872,0.001628757,0.00583563,0.000205173,0.00003602219,0.00001621626,0.0001433367,5.508373e-7,0.000008285479,0.06209269,0.006488166,0.923444],"study_design_scores_gemma":[0.002121972,0.0006678075,0.08158514,0.005164266,0.0004322662,0.001557483,0.003786171,0.06620632,0.0001795492,0.1120956,0.7256879,0.0005155065],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.04607846,0.003560723,0.04767673,0.8913528,0.009233855,0.0007514513,0.000003504627,0.00004808014,0.001294341],"genre_scores_gemma":[0.6853299,0.01747678,0.2234635,0.05260362,0.01922494,0.001021297,0.0004477585,0.00009585555,0.0003363265],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.9229285,"threshold_uncertainty_score":0.9961299,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04904060000455475,"score_gpt":0.4628714152828736,"score_spread":0.4138308152783189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}