{"id":"W4389348902","doi":"10.2196/49183","title":"ChatGPT Versus Consultants: Blinded Evaluation on Answering Otorhinolaryngology Case–Based Questions","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":43,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Otorhinolaryngology; Likert scale; Medicine; Medical education; German; Family medicine; Psychology; Surgery; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001091132,0.0001458353,0.0001938818,0.0004319514,0.0002332732,0.00002131829,0.00008482335,0.000322815,0.001412867],"category_scores_gemma":[0.008421555,0.0001395065,0.00005405767,0.001158124,0.0001187953,0.00009550974,0.00001362123,0.0004164612,0.0008774681],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004127432,"about_ca_system_score_gemma":0.005917348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005071749,"about_ca_topic_score_gemma":0.0001870841,"domain_scores_codex":[0.9978496,0.0001907879,0.0004642329,0.0003641366,0.0008052823,0.0003259998],"domain_scores_gemma":[0.997699,0.0005854992,0.0001119006,0.0003712607,0.0007850568,0.0004472695],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001111973,0.001880432,0.006007297,0.000251781,0.0000559958,0.0001816549,0.005069675,0.00008033619,0.0004444973,0.002387071,0.03077547,0.9517538],"study_design_scores_gemma":[0.008862256,0.008844328,0.253547,0.005618097,0.001221469,0.003167571,0.05300844,0.4892845,0.01333201,0.006474202,0.1541692,0.002470877],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9647722,0.0001005873,0.00009438397,0.02732633,0.004715455,0.001215255,0.000004007718,0.0002558446,0.001515908],"genre_scores_gemma":[0.9943923,0.00004666865,0.0001423934,0.00270272,0.001146678,0.0009712204,0.0003790032,0.00002332579,0.0001956604],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9492829,"threshold_uncertainty_score":0.9999309,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2135554631027048,"score_gpt":0.5264062817580218,"score_spread":0.312850818655317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}