{"id":"W4378976945","doi":"10.1101/2023.05.30.23290758","title":"Evaluating ChatGPT-4 in Otolaryngology–Head and Neck Surgery Board Examination using the CVSA Model","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Otorhinolaryngology; Concordance; Head and neck surgery; Medicine; Head and neck; SAFER; Sample (material); Consistency (knowledge bases); Medical physics; Multiple choice; Medical education; Surgery; Artificial intelligence; Computer science; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003414661,0.0001939042,0.0004116361,0.0003335479,0.0001412405,0.00003403759,0.0001035306,0.0003446418,0.00001417655],"category_scores_gemma":[0.00181476,0.0001580778,0.00007973242,0.0002680714,0.00009957789,0.00006188811,0.0001749377,0.0007786974,0.00001665254],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001913149,"about_ca_system_score_gemma":0.0006047729,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002905134,"about_ca_topic_score_gemma":0.0006746313,"domain_scores_codex":[0.9978918,0.0003141873,0.0006664083,0.0004820573,0.0003243529,0.0003212228],"domain_scores_gemma":[0.9982593,0.0007612595,0.0002315349,0.0004572023,0.0002043257,0.00008638501],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005107407,0.00006816453,0.9411045,0.0008214413,0.00002514658,0.0000238233,0.007418356,0.010121,0.001157682,0.00004954843,0.00008695288,0.03907226],"study_design_scores_gemma":[0.00002520253,0.00002707443,0.4535501,0.0008584919,0.00004777127,0.00001206091,0.0004983517,0.5415957,0.0005903437,0.002671045,0.0000094995,0.0001143581],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9926912,0.0005758879,0.0009744295,0.003677678,0.001132438,0.0008224195,0.000003290893,0.00006688701,0.00005578955],"genre_scores_gemma":[0.9971814,0.0004394181,0.001179076,0.0003317913,0.0004373573,0.0001475599,0.00004697427,0.00004037967,0.0001961142],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5314747,"threshold_uncertainty_score":0.6446227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4489275324866818,"score_gpt":0.4892135576655389,"score_spread":0.04028602517885704,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}