{"id":"W4410881775","doi":"10.2196/69521","title":"Chatbots’ Role in Generating Single Best Answer Questions for Undergraduate Medical Student Assessment: Comparative Analysis","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"AI in Service Interactions","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Chatbot; Consistency (knowledge bases); Computer science; Test (biology); Quality (philosophy); Data science; Scale (ratio); Medical education; Psychology; Artificial intelligence; Medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000570407,0.0001481466,0.000284021,0.0005017934,0.0001976048,0.000174437,0.0007940077,0.0001543273,0.0001418249],"category_scores_gemma":[0.0003204109,0.0001434274,0.0001078829,0.001757207,0.00006577999,0.0003898399,0.0001763196,0.0003666409,0.0000176768],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004184843,"about_ca_system_score_gemma":0.002251022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004583201,"about_ca_topic_score_gemma":0.003597872,"domain_scores_codex":[0.9977329,0.0002078239,0.0005363243,0.0004631671,0.0007984274,0.0002613509],"domain_scores_gemma":[0.9984428,0.0004866366,0.0001371075,0.000413198,0.0002732276,0.0002470238],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002448174,0.02323217,0.07531638,0.0001783947,0.001400123,0.00001286418,0.02276564,0.003455967,0.000806695,0.4378597,0.02015973,0.4147879],"study_design_scores_gemma":[0.0005672814,0.0001096778,0.02608928,0.0003632418,0.0001251351,0.000005116151,0.004353161,0.951749,0.00007694158,0.004843915,0.01146961,0.0002476231],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0791403,0.0001764406,0.8597814,0.05434957,0.001658266,0.0006246208,0.000002845954,0.00009729758,0.00416923],"genre_scores_gemma":[0.9653177,0.00001696782,0.02941976,0.003296173,0.0002078241,0.001055342,0.00006758857,0.000005556427,0.0006130978],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.948293,"threshold_uncertainty_score":0.5848801,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02190070761803816,"score_gpt":0.4410937310131128,"score_spread":0.4191930233950747,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}