{"id":"W4412744396","doi":"10.2196/69313","title":"Role of Artificial Intelligence in Surgical Training by Assessing GPT-4 and GPT-4o on the Japan Surgical Board Examination With Text-Only and Image-Accompanied Questions: Performance Evaluation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Workforce; Medicine; Medical education; Psychology; Artificial intelligence; Computer science; Political science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01065357,0.0008331314,0.0008844247,0.001648926,0.0004037536,0.001599358,0.0006893894,0.001160184,0.001758136],"category_scores_gemma":[0.0443042,0.0002422227,0.00140785,0.0009762778,0.0007391713,0.001641642,0.002110492,0.001010392,0.0008906285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009663887,"about_ca_system_score_gemma":0.001131086,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002835545,"about_ca_topic_score_gemma":0.002579207,"domain_scores_codex":[0.991057,0.004352606,0.00133322,0.001029482,0.001641784,0.0005858003],"domain_scores_gemma":[0.9563251,0.0272191,0.006285186,0.001774437,0.004679912,0.003716328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002761909,0.002558855,0.9026079,0.000381203,0.0003586299,0.0001332803,0.002227285,0.003796305,0.001577587,0.0001696855,0.001893779,0.08153361],"study_design_scores_gemma":[0.0002477594,0.004864512,0.9486163,0.0001257091,0.0003351204,0.0003089394,0.001814296,0.03599977,0.003893204,0.0004586615,0.003241837,0.0000938714],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9965131,0.0001521074,0.0007072412,0.0001296247,0.00002371431,0.0001729764,0.0002614762,0.00006607712,0.001973691],"genre_scores_gemma":[0.9958727,0.0001421871,0.002375394,0.00007526451,0.0000234319,0.000219959,0.0006484405,0.00001377033,0.0006290219],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01065357,"threshold_uncertainty_score":0.05634212,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07409362411161255,"score_gpt":0.4485447978764953,"score_spread":0.3744511737648827,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}