{"id":"W3108940771","doi":"10.4018/978-1-7998-3473-1.ch019","title":"Automating the Generation of Test Items","year":2020,"lang":"en","type":"book-chapter","venue":"Advances in logistics, operations, and management science book series","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Test (biology); Computer science; Domain (mathematical analysis); Data science; Machine learning; Artificial intelligence; Information retrieval; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0007823466,0.0001772868,0.0002222662,0.0001403789,0.001129872,0.0003120285,0.0004828892,0.000060096,0.00007016731],"category_scores_gemma":[0.0002939786,0.0001446811,0.0000315834,0.0001503547,0.002944719,0.001695406,0.0002178945,0.0001273781,0.000009476968],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009163886,"about_ca_system_score_gemma":0.000124859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000541273,"about_ca_topic_score_gemma":0.002157522,"domain_scores_codex":[0.9983163,0.00002747809,0.0004153036,0.0003913204,0.0006333097,0.0002163097],"domain_scores_gemma":[0.9993272,0.0001182693,0.0001698295,0.0002048546,0.0001267666,0.00005310135],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000002225867,0.00001059039,0.0004635323,0.00004657165,0.000007480287,0.000002904715,0.001811293,0.0005333659,0.00002752402,0.9930251,0.0003328818,0.003736566],"study_design_scores_gemma":[0.0002754181,0.0001731749,0.0009921371,0.0002595116,0.00009379262,9.061554e-7,0.007616338,0.002507817,0.00004349826,0.01990968,0.9675831,0.0005445864],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00002590791,0.005866128,0.002576508,0.002679468,0.0007571763,0.001037599,0.00001960254,0.00005724392,0.9869804],"genre_scores_gemma":[0.08852676,0.3302948,0.01807881,0.001695139,0.001276666,0.0001688669,0.0000771027,0.00004762354,0.5598342],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.9731154,"threshold_uncertainty_score":0.9997687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04221560599566671,"score_gpt":0.3313379888908244,"score_spread":0.2891223828951577,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}