{"id":"W4412591596","doi":"10.1007/978-3-031-99267-4_11","title":"An Augmented Intelligence System for Automated Quality Control and Feedback Generation of Multiple Choice Test Items","year":2025,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Test (biology); Multiple choice; Computer science; Control (management); Quality (philosophy); Artificial intelligence; Biology; Statistics; Mathematics; Botany","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008418647,0.0009883345,0.0009615311,0.001245274,0.0004297858,0.001465744,0.002064076,0.0009078549,0.02342939],"category_scores_gemma":[0.002367527,0.000652936,0.0004037545,0.000943412,0.0003148615,0.001006588,0.001131282,0.0007303178,0.005569703],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004437706,"about_ca_system_score_gemma":0.000880483,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003720284,"about_ca_topic_score_gemma":0.003709541,"domain_scores_codex":[0.9991077,0.0001402218,0.00007277852,0.000227995,0.000406108,0.000045179],"domain_scores_gemma":[0.9982298,0.0007474305,0.0001087306,0.0003082244,0.0005107386,0.00009501947],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00261293,0.0004926588,0.002297753,0.0002730406,0.00009922496,0.000308127,0.0003542223,0.005247458,0.09670122,0.002559807,0.02873461,0.860319],"study_design_scores_gemma":[0.001027772,0.002426804,0.02104109,0.0001689645,0.00079459,0.002503275,0.0001960013,0.6257415,0.1996867,0.007171199,0.1388503,0.0003918133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04166984,0.000397749,0.8551459,0.000138945,0.0003130028,0.0008019301,0.001553471,0.08608609,0.01389306],"genre_scores_gemma":[0.2105313,0.0002953465,0.7541202,0.0004717871,0.0001576118,0.001361641,0.002304844,0.001726038,0.02903129],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02342939,"threshold_uncertainty_score":0.07837915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07028701809007214,"score_gpt":0.3364495307656157,"score_spread":0.2661625126755435,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}