{"id":"W4410424798","doi":"10.2196/66917","title":"Benchmarking the Confidence of Large Language Models in Answering Clinical Questions: Cross-Sectional Evaluation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Confidence interval; Low Confidence; Medicine; Statistics; Benchmarking; Mean difference; Psychology; Internal medicine; Mathematics; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00578257,0.0000752583,0.0002005054,0.0001396507,0.00009516934,0.00002316099,0.0001440634,0.0001643554,0.000217308],"category_scores_gemma":[0.00218069,0.00005405941,0.00005527904,0.0003791982,0.0001378579,0.0002058145,0.00005463593,0.0005005922,0.000009212783],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001052478,"about_ca_system_score_gemma":0.0009405491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003207617,"about_ca_topic_score_gemma":0.0003367594,"domain_scores_codex":[0.9973098,0.0001510872,0.001455333,0.00009436108,0.0008157072,0.0001737373],"domain_scores_gemma":[0.9983805,0.0007376425,0.0001824689,0.0002572374,0.0003585092,0.00008363718],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00008307816,0.0007492368,0.8971097,0.0002253374,0.0000320317,0.000002673452,0.02634818,0.0004286128,0.000002941829,0.00299004,0.0002201454,0.071808],"study_design_scores_gemma":[0.0003393347,0.0002002816,0.525911,0.0004699012,0.0000272785,0.000006779741,0.03281957,0.4382174,0.00004690636,0.00177321,0.0001288599,0.00005947513],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9908679,0.0001429751,0.005328432,0.0004445837,0.0006359189,0.001043389,0.000002185334,0.00001960435,0.001515021],"genre_scores_gemma":[0.9984114,0.00006286235,0.0002420737,0.0009129566,0.0001679455,0.0001389266,0.00002047114,0.000003276912,0.00004014638],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4377888,"threshold_uncertainty_score":0.2610647,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1903462553477343,"score_gpt":0.5662406282886161,"score_spread":0.3758943729408818,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}