{"id":"W4416618182","doi":"10.2196/79465","title":"The Validity of Generative Artificial Intelligence in Evaluating Medical Students in Objective Structured Clinical Examination: Experimental Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Clinical trial; Artificial neural network; Medical information; Medical research; MEDLINE; Reliability (semiconductor)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03490905,0.0006605936,0.0006417368,0.001015036,0.00079109,0.001502041,0.0009840674,0.001241998,0.001146811],"category_scores_gemma":[0.1018287,0.0005326312,0.001049091,0.0005882678,0.002559219,0.001881299,0.002207123,0.001016713,0.000310585],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001000233,"about_ca_system_score_gemma":0.001267649,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006237095,"about_ca_topic_score_gemma":0.0008512682,"domain_scores_codex":[0.9713618,0.01951837,0.001965355,0.002328423,0.004009336,0.0008168175],"domain_scores_gemma":[0.8639466,0.09654715,0.01461602,0.01114606,0.009948871,0.003795302],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01682975,0.04048905,0.7886052,0.0006335761,0.0005859457,0.0003140464,0.02653855,0.004221171,0.01064225,0.001689913,0.0005252834,0.1089253],"study_design_scores_gemma":[0.004330128,0.1279452,0.7846954,0.0002693263,0.0005909925,0.0006251827,0.00889259,0.05116613,0.01519627,0.003154967,0.002811984,0.0003218269],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9977232,0.00001654444,0.001115191,0.00002499873,0.00001319149,0.0004995132,0.00001281624,0.000006557517,0.0005880404],"genre_scores_gemma":[0.9925491,0.00002788251,0.00595497,0.00007747292,0.00002620539,0.00112144,0.00004535318,0.000005289624,0.0001923921],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9650909,"threshold_uncertainty_score":0.1846189,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5015174914828086,"score_gpt":0.6725896981966943,"score_spread":0.1710722067138858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}