{"id":"W4400642441","doi":"10.1016/j.acra.2024.06.046","title":"Large Language Models as Tools to Generate Radiology Board-Style Multiple-Choice Questions","year":2024,"lang":"en","type":"article","venue":"Academic Radiology","topic":"Radiology practices and education","field":"Medicine","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan; Royal University Hospital","funders":"","keywords":"Style (visual arts); Radiology; Computer science; Editorial board; Medicine; Library science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0461589,0.001669248,0.0006178896,0.00251901,0.0007407692,0.004009455,0.00187861,0.001097249,0.01602177],"category_scores_gemma":[0.1918933,0.0008350572,0.00123589,0.001252067,0.0008854678,0.003643791,0.004877694,0.002205552,0.003232999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002402609,"about_ca_system_score_gemma":0.003440931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001073351,"about_ca_topic_score_gemma":0.002452739,"domain_scores_codex":[0.9529658,0.03873483,0.002280463,0.001800191,0.003626825,0.0005918958],"domain_scores_gemma":[0.7075827,0.2547524,0.01012358,0.009817723,0.01579157,0.001931999],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002968708,0.00343581,0.02859026,0.003018259,0.0002165662,0.0007027822,0.02223433,0.03065692,0.01716522,0.02920279,0.05994131,0.801867],"study_design_scores_gemma":[0.003526419,0.004451976,0.02450841,0.004303554,0.0004939247,0.001352654,0.01680665,0.531562,0.05737328,0.09573889,0.258905,0.0009772124],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1741181,0.000562818,0.7715143,0.003166357,0.0005170124,0.01337223,0.004181729,0.01301112,0.01955635],"genre_scores_gemma":[0.2350532,0.0001995433,0.745499,0.0006074469,0.00008987785,0.0115376,0.002741056,0.0008216465,0.003450643],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0461589,"threshold_uncertainty_score":0.2441145,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04522509973484705,"score_gpt":0.374001664389616,"score_spread":0.328776564654769,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}