{"id":"W4389925007","doi":"10.21449/ijate.1359348","title":"Automatic item generation for non-verbal reasoning items","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Verbal reasoning; Computer science; Test (biology); Item analysis; Cognition; Item response theory; Homogeneous; Subject-matter expert; Natural language processing; Artificial intelligence; Subject matter; Psychology; Cognitive psychology; Psychometrics; Pedagogy; Expert system; Mathematics; Developmental psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02336156,0.001758417,0.001543125,0.004638094,0.0005493676,0.001595341,0.002544604,0.001026058,0.005750282],"category_scores_gemma":[0.09864897,0.0006700027,0.001638967,0.003432713,0.0006237978,0.001540012,0.001850274,0.0012339,0.004555668],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007100459,"about_ca_system_score_gemma":0.001512976,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004610643,"about_ca_topic_score_gemma":0.0006673317,"domain_scores_codex":[0.9789555,0.01287478,0.002438391,0.001753783,0.003670365,0.0003072084],"domain_scores_gemma":[0.8852344,0.07012389,0.005273379,0.01595028,0.02286717,0.00055089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00115536,0.001199746,0.03443938,0.001251562,0.0002016239,0.0003929079,0.002044761,0.005760952,0.01713522,0.003500544,0.01083002,0.9220879],"study_design_scores_gemma":[0.003003121,0.009654179,0.2186646,0.001933947,0.001087248,0.005309829,0.006242669,0.4304003,0.1903456,0.04424092,0.08819952,0.0009181393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1955926,0.0003296204,0.7729322,0.0002582302,0.000338698,0.01394805,0.002893276,0.008202627,0.005504637],"genre_scores_gemma":[0.1867292,0.000174963,0.7894253,0.0001523482,0.0000638622,0.01415372,0.006707145,0.0006653521,0.001928042],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02336156,"threshold_uncertainty_score":0.1235492,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08817372903914804,"score_gpt":0.4972479358557301,"score_spread":0.4090742068165821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}