{"id":"W4414645151","doi":"10.2196/70190","title":"Enhancing Large Language Models for Improved Accuracy and Safety in Medical Question Answering: Comparative Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretability; Reliability (semiconductor); Benchmark (surveying); Key (lock); Medical information; Medical decision making; Medical practice; MEDLINE","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001372238,0.0001193804,0.0002857211,0.0001976548,0.0001094312,0.00001816879,0.00008483633,0.0002119888,0.000142352],"category_scores_gemma":[0.00304687,0.0001070763,0.00003185579,0.0003091121,0.00005385426,0.0001565845,0.00003228113,0.0003535055,0.000003828828],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002081208,"about_ca_system_score_gemma":0.00287169,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009126461,"about_ca_topic_score_gemma":0.002602794,"domain_scores_codex":[0.9984283,0.0001176825,0.0005862153,0.0003153494,0.0003063623,0.0002461206],"domain_scores_gemma":[0.9987391,0.000518797,0.00008104097,0.000182231,0.0001935614,0.0002852298],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001609144,0.009694094,0.03058454,0.001128039,0.0000797211,0.000007814982,0.1408128,0.000006484619,0.0009412106,0.01037268,0.003091529,0.8016719],"study_design_scores_gemma":[0.004554079,0.00230615,0.1409433,0.007162786,0.000236114,0.00006078018,0.5083036,0.3002037,0.007580233,0.01542027,0.01234357,0.0008853539],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9654535,0.0005349144,0.01322593,0.01699933,0.0007788729,0.002488052,0.000002458468,0.00004676687,0.0004702029],"genre_scores_gemma":[0.9956229,0.00009836857,0.000367342,0.002424956,0.0003422987,0.0007654264,0.00009291635,0.000008475644,0.0002772576],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8007866,"threshold_uncertainty_score":0.5094255,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06302532221165427,"score_gpt":0.5048432072102752,"score_spread":0.441817884998621,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}