{"id":"W4386117191","doi":"10.1111/medu.15190","title":"Validity evidence supporting clinical skills assessment by artificial intelligence compared with trained clinician raters","year":2023,"lang":"en","type":"article","venue":"Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Wilson Centre; University of Toronto","funders":"Novo Nordisk Fonden; Novo Nordisk","keywords":"Psychology; Educational measurement; MEDLINE; Medical education; Clinical psychology; Applied psychology; Medicine; Curriculum; Pedagogy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006976562,0.0002129792,0.0004679109,0.0001749942,0.0002348359,0.00005990117,0.0002375175,0.0002824426,0.001145933],"category_scores_gemma":[0.01023615,0.0001749827,0.000124657,0.0009346742,0.0003530946,0.0002020026,0.00003773082,0.0007248917,0.0004321669],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002621538,"about_ca_system_score_gemma":0.007515195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004690139,"about_ca_topic_score_gemma":0.0001681494,"domain_scores_codex":[0.995295,0.0005229725,0.001670531,0.0006214947,0.001336242,0.000553734],"domain_scores_gemma":[0.9962243,0.001581102,0.0003968461,0.0004709305,0.0006331975,0.0006935855],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0002053511,0.002039197,0.1295921,0.0001525272,0.00003586154,0.00001581624,0.001994038,0.00001564299,0.0001077047,0.000224859,0.0571837,0.8084332],"study_design_scores_gemma":[0.001011888,0.00980048,0.6047404,0.01199122,0.001747996,0.0003692966,0.1045564,0.1541166,0.02653332,0.02357217,0.05829539,0.003264793],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9382365,0.0000520306,0.006202666,0.05121397,0.002893326,0.0007801432,0.000004550493,0.0002230694,0.0003937489],"genre_scores_gemma":[0.9902046,0.0002898354,0.001029994,0.006108005,0.001440609,0.0001531678,0.000339904,0.00003063077,0.0004032419],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8051684,"threshold_uncertainty_score":0.9997672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3557575284167163,"score_gpt":0.5833110180889678,"score_spread":0.2275534896722515,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}