{"id":"W4394580931","doi":"10.35542/osf.io/6d8tj","title":"A Review of Automatic Item Generation Techniques Leveraging Large Language Models","year":2024,"lang":"en","type":"review","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Language model; Data science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02196107,0.002706224,0.003200324,0.0132544,0.0006696316,0.002864552,0.003990593,0.001499741,0.005718126],"category_scores_gemma":[0.07223466,0.001505438,0.004238543,0.01201708,0.0009740136,0.004382811,0.001447397,0.001801185,0.003769623],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001473964,"about_ca_system_score_gemma":0.004540528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005109525,"about_ca_topic_score_gemma":0.007448732,"domain_scores_codex":[0.986156,0.007586812,0.002254439,0.001245276,0.002595461,0.0001620339],"domain_scores_gemma":[0.8937407,0.09393409,0.003161629,0.002427419,0.006479062,0.0002570976],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001086217,0.0001214132,0.001723102,0.04539774,0.0007277231,0.0001115662,0.000412437,0.001944195,0.0006736085,0.002545685,0.009829079,0.9364049],"study_design_scores_gemma":[0.000443932,0.001592979,0.04521264,0.1622936,0.009971125,0.003355216,0.002316582,0.04289169,0.009801196,0.03708444,0.6841079,0.000928692],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.003092591,0.9032176,0.08433232,0.001820466,0.0003782375,0.0007929968,0.001461229,0.001126576,0.00377791],"genre_scores_gemma":[0.02648623,0.7300932,0.2344056,0.001483567,0.0006391485,0.001935094,0.003326217,0.0003396182,0.001291231],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.02196107,"threshold_uncertainty_score":0.1161427,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1025841590571828,"score_gpt":0.3687607373528912,"score_spread":0.2661765782957084,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}