{"id":"W4388478197","doi":"10.2196/49970","title":"A Novel Evaluation Model for Assessing ChatGPT on Otolaryngology–Head and Neck Surgery Certification Examinations: Performance Study","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Otorhinolaryngology; Head and neck surgery; Certification; Medicine; Head and neck; Rhinology; Head (geology); Medical physics; General surgery; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0234163,0.001427402,0.0009726959,0.002519978,0.0006014758,0.002633789,0.001381048,0.001234009,0.002086449],"category_scores_gemma":[0.06356451,0.0002878996,0.00163492,0.001338582,0.0009128129,0.002432838,0.001636865,0.001156339,0.000493744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003503549,"about_ca_system_score_gemma":0.003154125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01014041,"about_ca_topic_score_gemma":0.007104628,"domain_scores_codex":[0.9864193,0.00788306,0.001048817,0.001571327,0.002591186,0.0004864065],"domain_scores_gemma":[0.9204078,0.05839984,0.004229472,0.003107548,0.01263552,0.001219913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00320115,0.003009559,0.51775,0.0008729718,0.0009807741,0.0002455646,0.003150325,0.1234587,0.005794223,0.007111679,0.004260036,0.3301649],"study_design_scores_gemma":[0.0001388925,0.002403857,0.05710479,0.00009700633,0.0002560072,0.0001378825,0.0006650992,0.9331964,0.002891402,0.001822154,0.001214119,0.00007225686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6781818,0.0002975178,0.3078701,0.000649259,0.00009107178,0.002845673,0.0008080013,0.000945518,0.008311169],"genre_scores_gemma":[0.9142134,0.00006809294,0.08286325,0.000066197,0.00002458507,0.0012055,0.0006934142,0.00003050932,0.0008350168],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0234163,"threshold_uncertainty_score":0.1238387,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3351334086912761,"score_gpt":0.5184164862577221,"score_spread":0.183283077566446,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}