{"id":"W4392702964","doi":"10.2196/54393","title":"Capability of GPT-4V(ision) in the Japanese National Medical Licensing Examination: Evaluation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":59,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"McNemar's test; Test (biology); Artificial intelligence; Computer science; Medical physics; Medicine; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01034817,0.0001220516,0.0002030399,0.0003060577,0.00008413241,0.0000331129,0.0001827814,0.0002392605,0.00271607],"category_scores_gemma":[0.01185174,0.00008614005,0.00006385281,0.001021603,0.0001610076,0.0001833588,0.00002250194,0.0005603274,0.00006434347],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004656115,"about_ca_system_score_gemma":0.008440555,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008909063,"about_ca_topic_score_gemma":0.000451269,"domain_scores_codex":[0.9939867,0.0007411122,0.0008437529,0.0003600059,0.003883827,0.0001846437],"domain_scores_gemma":[0.9973158,0.001101966,0.00009333239,0.0002948203,0.0009728072,0.0002213106],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00005626178,0.005278693,0.02757193,0.0004260705,0.00002746032,0.000009759407,0.09349848,0.000004493624,0.00003175692,0.001177932,0.006717059,0.8652001],"study_design_scores_gemma":[0.0003475994,0.0006812108,0.837108,0.001782211,0.0001366163,0.0002168755,0.1113291,0.0399208,0.0001555871,0.004268794,0.003827173,0.0002260192],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.960896,0.0005621546,0.00002447051,0.03247608,0.001682248,0.001636931,9.490084e-7,0.0000346188,0.002686562],"genre_scores_gemma":[0.9964956,0.00003259048,0.00003577431,0.001532681,0.001207536,0.000498935,0.00009774121,0.00001169626,0.00008745591],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8649741,"threshold_uncertainty_score":0.9981956,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1563856543944998,"score_gpt":0.5352369858648602,"score_spread":0.3788513314703604,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}