{"id":"W4409155725","doi":"10.2196/65726","title":"Using a Hybrid of AI and Template-Based Method in Automatic Item Generation to Create Multiple-Choice Questions in Medical Education: Hybrid AIG","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Summative assessment; Formative assessment; Field (mathematics); Artificial intelligence; Template; Subject-matter expert; Machine learning; Expert system; Programming language; Psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02454995,0.002018762,0.001072052,0.003785824,0.0007617453,0.002619061,0.003495169,0.002063296,0.006103813],"category_scores_gemma":[0.06845504,0.001068287,0.001720727,0.002596337,0.001646283,0.003571581,0.002951531,0.001624409,0.002376514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001433369,"about_ca_system_score_gemma":0.001487215,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008016148,"about_ca_topic_score_gemma":0.001329071,"domain_scores_codex":[0.973289,0.0192092,0.001349999,0.002953406,0.002891029,0.0003075659],"domain_scores_gemma":[0.8884308,0.09235949,0.002998057,0.008750647,0.006686624,0.0007743936],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006530461,0.001383683,0.008655904,0.001699587,0.0003370377,0.0004333282,0.01040617,0.01716245,0.02636399,0.01514876,0.003812101,0.9139438],"study_design_scores_gemma":[0.001530667,0.003984055,0.01752669,0.0009361862,0.0005397559,0.002289431,0.004402829,0.7445952,0.08826011,0.08637533,0.04876757,0.0007922167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01863009,0.00009759536,0.9749789,0.0001591831,0.00004698386,0.002011307,0.0001163936,0.002099763,0.001859904],"genre_scores_gemma":[0.04170315,0.00003192988,0.9555932,0.00007951474,0.00001087349,0.00188114,0.0001292722,0.000107756,0.0004630821],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02454995,"threshold_uncertainty_score":0.1298341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5171630694068262,"score_gpt":0.6303167085209163,"score_spread":0.11315363911409,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}