{"id":"W4395018673","doi":"10.2196/54283","title":"The Performance of ChatGPT-4V in Interpreting Images and Tables in the Japanese Medical Licensing Exam","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Table (database); Computer science; Data mining; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00231698,0.00007799624,0.0001302202,0.000122525,0.00007251544,0.00003542594,0.0001262885,0.0001353056,0.0001443555],"category_scores_gemma":[0.002876408,0.00004418433,0.00002194093,0.0004128409,0.0002397935,0.0001100371,0.0000260633,0.0005293885,0.000008151874],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007442155,"about_ca_system_score_gemma":0.001399107,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001213562,"about_ca_topic_score_gemma":0.0003967119,"domain_scores_codex":[0.9984949,0.0001355907,0.0004620061,0.0001693685,0.0005454969,0.0001926446],"domain_scores_gemma":[0.9985697,0.001033953,0.00004731676,0.0001612365,0.00006786966,0.0001199865],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00005646863,0.000241987,0.04890443,0.0005926897,0.000005922614,0.000007941488,0.06126044,5.066947e-7,0.0001050381,0.000373501,0.002790123,0.8856609],"study_design_scores_gemma":[0.0003812035,0.001032157,0.4389764,0.03389378,0.00008723018,0.001180259,0.3481864,0.1192576,0.004508038,0.0018618,0.05007617,0.0005589679],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9380873,0.003837814,0.000004133226,0.05639032,0.0007290097,0.0003149158,2.39612e-7,0.0000130158,0.0006232808],"genre_scores_gemma":[0.9960425,0.001731786,0.00002173301,0.001524223,0.0004564005,0.0001028765,0.000008699316,0.000008112534,0.0001036769],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.885102,"threshold_uncertainty_score":0.3443537,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03461135957563634,"score_gpt":0.4288645519000784,"score_spread":0.3942531923244421,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}