{"id":"W4388478197","doi":"10.2196/49970","title":"A Novel Evaluation Model for Assessing ChatGPT on Otolaryngology–Head and Neck Surgery Certification Examinations: Performance Study","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Otorhinolaryngology; Head and neck surgery; Certification; Medicine; Head and neck; Rhinology; Head (geology); Medical physics; General surgery; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002392847,0.0001451039,0.0002308058,0.0004642663,0.0003110188,0.00004655609,0.00007026067,0.0002063713,0.00005403307],"category_scores_gemma":[0.005012028,0.0001369093,0.0000474182,0.0005955051,0.00007046897,0.0002922565,0.0000158455,0.0002274584,0.00004554064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002469026,"about_ca_system_score_gemma":0.002867864,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007685619,"about_ca_topic_score_gemma":0.00003409682,"domain_scores_codex":[0.9978917,0.0000913296,0.0005702379,0.0004338098,0.000741376,0.0002715056],"domain_scores_gemma":[0.9979445,0.0005941345,0.000171377,0.0003010996,0.0007429613,0.0002459232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001368216,0.002930656,0.1170025,0.000419771,0.00002169064,3.899172e-7,0.009698423,0.000101368,0.0003206851,0.0001841928,0.004497392,0.8646861],"study_design_scores_gemma":[0.0001467647,0.00016179,0.518477,0.0003216721,0.00004581147,0.000008611474,0.002908911,0.4772558,0.0001628318,0.0002614514,0.0001557837,0.0000936237],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9855437,0.00006424352,0.001550129,0.008651867,0.001044855,0.002913006,0.000002142857,0.0001164354,0.0001135871],"genre_scores_gemma":[0.9928409,0.00008948342,0.0004279456,0.001068454,0.0006491583,0.004171477,0.0004518421,0.00002609433,0.0002746478],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8645925,"threshold_uncertainty_score":0.6000227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3351334086912761,"score_gpt":0.5184164862577221,"score_spread":0.183283077566446,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}