{"id":"W4389348902","doi":"10.2196/49183","title":"ChatGPT Versus Consultants: Blinded Evaluation on Answering Otorhinolaryngology Case–Based Questions","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":43,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Otorhinolaryngology; Likert scale; Medicine; Medical education; German; Family medicine; Psychology; Surgery; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06670874,0.0008363872,0.001523775,0.001514065,0.0009069709,0.001357413,0.0007696519,0.001705248,0.004290473],"category_scores_gemma":[0.1909064,0.0004798577,0.001165211,0.0009176719,0.001404284,0.001143889,0.00340896,0.0007793319,0.001609052],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001050826,"about_ca_system_score_gemma":0.001107155,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007449889,"about_ca_topic_score_gemma":0.000882838,"domain_scores_codex":[0.9249344,0.0524733,0.009167405,0.005034347,0.006837716,0.001552865],"domain_scores_gemma":[0.7000932,0.2134314,0.02830224,0.0167564,0.03465115,0.006765574],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.07463489,0.004958566,0.6261187,0.002748891,0.0009531843,0.002117433,0.05480055,0.002457795,0.03585194,0.0004929093,0.004343172,0.1905221],"study_design_scores_gemma":[0.005106741,0.04964298,0.8592299,0.0005590254,0.001279008,0.00353594,0.01702166,0.01662289,0.03515737,0.001080062,0.01021425,0.0005502516],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9932351,0.000191094,0.00248942,0.0001194107,0.0001012558,0.001721569,0.0002667572,0.00008499088,0.001790492],"genre_scores_gemma":[0.990826,0.0001185642,0.00504588,0.0002129487,0.0001131669,0.002569938,0.0003581301,0.00004477567,0.0007105726],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9332913,"threshold_uncertainty_score":0.3527938,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2135554631027048,"score_gpt":0.5264062817580218,"score_spread":0.312850818655317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}