{"id":"W4383263628","doi":"10.2196/50336","title":"Authors’ Reply to: Variability in Large Language Models’ Responses to Medical Licensing and Certification Examinations","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences","keywords":"Certification; Medicine; Psychology; Linguistics; Medical education; Medical physics; Political science; Philosophy; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004486568,0.0001221061,0.0002094486,0.0005574985,0.0001339153,0.00002965959,0.0001172465,0.0002612362,0.0005393552],"category_scores_gemma":[0.03393801,0.0001184435,0.0000297792,0.001476203,0.00005777721,0.0001391072,0.00006572462,0.0003817299,0.0002075934],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002687237,"about_ca_system_score_gemma":0.002839683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001239737,"about_ca_topic_score_gemma":0.0004588305,"domain_scores_codex":[0.997363,0.0003472042,0.0006178971,0.0004846683,0.0008266487,0.0003605213],"domain_scores_gemma":[0.9972069,0.0007914181,0.00006003411,0.0004176048,0.0002628071,0.001261278],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0003072989,0.001563034,0.04573558,0.0002983926,0.00001123394,0.00002257704,0.09515645,0.000007884021,0.0007794485,0.004528248,0.03845246,0.8131374],"study_design_scores_gemma":[0.0002709697,0.0003685207,0.8681836,0.001673765,0.00003729592,0.00009445357,0.03864893,0.05547659,0.001499739,0.005655931,0.02762284,0.0004673915],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.826368,0.00003680991,0.001474342,0.1699478,0.0006952455,0.0008820148,0.000003594569,0.000129074,0.000463045],"genre_scores_gemma":[0.9838363,0.00004597467,0.000637448,0.01330805,0.0005413637,0.0004763221,0.0001275139,0.00001995613,0.001007089],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.822448,"threshold_uncertainty_score":0.9741995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1278122145063649,"score_gpt":0.4869630424352007,"score_spread":0.3591508279288358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}