{"id":"W4381376804","doi":"10.2196/preprints.49970","title":"A Novel Evaluation Model for Assessing ChatGPT on Otolaryngology–Head and Neck Surgery Certification Examinations: Performance Study (Preprint)","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Concordance; Certification; Otorhinolaryngology; Head and neck surgery; Board certification; Sample (material); Medicine; Head and neck; Consistency (knowledge bases); Multiple choice; Medical education; Medical physics; Test (biology); Computer science; Artificial intelligence; Surgery; Continuing medical education","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01762881,0.001412712,0.001022785,0.001806501,0.0004797948,0.002372579,0.001250819,0.001200563,0.003090977],"category_scores_gemma":[0.04800657,0.0002904526,0.00129785,0.001138165,0.0006628285,0.001952115,0.001347801,0.001095183,0.000737567],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00335865,"about_ca_system_score_gemma":0.002936449,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0141246,"about_ca_topic_score_gemma":0.008271559,"domain_scores_codex":[0.9913762,0.005013637,0.0006208356,0.00122888,0.001435771,0.0003246448],"domain_scores_gemma":[0.9452781,0.04171529,0.002330364,0.002011815,0.007867369,0.0007969765],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00461127,0.003615398,0.3190711,0.0009804697,0.0009415952,0.0002608121,0.002010849,0.2222699,0.007257442,0.007076804,0.007508072,0.4243963],"study_design_scores_gemma":[0.00009587059,0.001236005,0.02326359,0.00004614909,0.0001704457,0.00006310426,0.0002599619,0.9707677,0.002325386,0.001042835,0.0006853186,0.00004370504],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6386307,0.0003426585,0.3475583,0.0007966019,0.0001196375,0.002637676,0.00138069,0.001463086,0.007070645],"genre_scores_gemma":[0.8901678,0.00008990627,0.1052626,0.0001060912,0.00003177079,0.0015704,0.00123795,0.000047809,0.001485686],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01762881,"threshold_uncertainty_score":0.09323114,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.551301037349342,"score_gpt":0.5006886983653596,"score_spread":0.05061233898398243,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}