{"id":"W4383263628","doi":"10.2196/50336","title":"Authors’ Reply to: Variability in Large Language Models’ Responses to Medical Licensing and Certification Examinations","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences","keywords":"Certification; Medicine; Psychology; Linguistics; Medical education; Medical physics; Political science; Philosophy; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007319777,0.0005662984,0.001116424,0.001059882,0.002780946,0.003335171,0.001396617,0.02582256,0.01214592],"category_scores_gemma":[0.1246547,0.0007220085,0.0009708705,0.0009119479,0.002387358,0.002273679,0.002235173,0.0196765,0.008203053],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002971557,"about_ca_system_score_gemma":0.003674647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005837271,"about_ca_topic_score_gemma":0.006028948,"domain_scores_codex":[0.9946839,0.001553148,0.001252005,0.000815078,0.001095512,0.0006003244],"domain_scores_gemma":[0.9386154,0.04075756,0.003072872,0.002031476,0.01341417,0.002108472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001478605,0.000018784,0.002069091,0.000112074,0.00004692639,0.0007033152,0.0007517588,0.0001025464,0.0002309695,0.0007744654,0.9900026,0.005039579],"study_design_scores_gemma":[0.000238085,0.00009923759,0.007719499,0.0007953842,0.0001868472,0.003272152,0.004446111,0.001249305,0.001299262,0.006276105,0.9741662,0.000251922],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.001217074,0.000518447,0.0002373604,0.9435003,0.053558,0.000009882361,0.0002964381,0.00005110146,0.0006114746],"genre_scores_gemma":[0.01785306,0.0006693001,0.0004968471,0.9299881,0.04656709,0.00006716746,0.0001972989,0.0000880305,0.004073148],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.02582256,"threshold_uncertainty_score":0.04063219,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1278122145063649,"score_gpt":0.4869630424352007,"score_spread":0.3591508279288358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}