{"id":"W4407421534","doi":"10.2196/71664","title":"Correction: Evaluating Bard Gemini Pro and GPT-4 Vision Against Student Performance in Medical Visual Question Answering: Comparative Case Study","year":2025,"lang":"en","type":"erratum","venue":"JMIR Formative Research","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Information retrieval; Computer science; Medical physics; Optometry; Medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004828821,0.001716315,0.002372756,0.003942303,0.004067359,0.003883132,0.00460826,0.01439296,0.03400441],"category_scores_gemma":[0.1187114,0.001280311,0.001463665,0.002148824,0.004196147,0.00222622,0.002165514,0.01425972,0.02706472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005907041,"about_ca_system_score_gemma":0.006399993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04443412,"about_ca_topic_score_gemma":0.04353879,"domain_scores_codex":[0.9932883,0.0009206186,0.001487276,0.0007273843,0.003082262,0.0004940866],"domain_scores_gemma":[0.9255732,0.01729234,0.003213126,0.003398249,0.04834049,0.00218257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00003750186,0.000009308158,0.00007393226,0.0001169219,0.00000918814,0.0004425925,0.00006792082,0.00002733814,0.00005177493,0.0003793575,0.9944956,0.004288492],"study_design_scores_gemma":[0.0001106558,0.0000646978,0.00207606,0.0007976312,0.0000782668,0.00195533,0.0005627677,0.0006536394,0.000757215,0.001576348,0.9912657,0.0001016472],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"editorial","genre_gemma":"other","genre_scores_codex":[0.000316492,0.0006871157,0.0006593442,0.1075175,0.8876984,0.00004589551,0.0009222765,0.000384928,0.001768126],"genre_scores_gemma":[0.03297378,0.006076286,0.006989598,0.2815758,0.4417158,0.0006188122,0.001747853,0.002252375,0.2260497],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.04443412,"threshold_uncertainty_score":0.1137561,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09632221086965453,"score_gpt":0.505090904754585,"score_spread":0.4087686938849304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}