{"id":"W2460844011","doi":"10.7202/1036327ar","title":"A Methodology for Multilingual Automatic Item Generation","year":2016,"lang":"en","type":"article","venue":"Mesure et évaluation en éducation","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Item bank; Task (project management); Test (biology); Process (computing); Quality (philosophy); Item response theory; Natural language processing; Multilingualism; Artificial intelligence; Information retrieval; Psychometrics; Psychology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008696053,0.0001135495,0.00012657,0.0001807748,0.0001270514,0.0000431828,0.0003125307,0.0001302479,0.00004484236],"category_scores_gemma":[0.002922812,0.00008673104,0.00005221436,0.0001957381,0.0000255328,0.0004980147,0.0000370929,0.00006196101,0.00004420254],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002029639,"about_ca_system_score_gemma":0.0007055813,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004892232,"about_ca_topic_score_gemma":0.00002767281,"domain_scores_codex":[0.9974484,0.001492648,0.0003064512,0.0003656929,0.0002248693,0.0001619078],"domain_scores_gemma":[0.9967736,0.002319828,0.0001867062,0.0003740792,0.0003140639,0.00003171733],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.000004274534,0.0002194096,0.001620149,0.00002289953,0.00004614473,1.934016e-7,0.01738892,0.00007759792,0.08762998,0.3119933,0.005385045,0.5756121],"study_design_scores_gemma":[0.002523919,0.0003957448,0.3415445,0.0001072524,0.00009464085,0.00003925895,0.0006325422,0.251779,0.1798087,0.2120489,0.01023632,0.0007891589],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2087915,0.00002823246,0.7579012,0.03145651,0.00116601,0.0003807612,0.000002512665,0.0001793975,0.00009389027],"genre_scores_gemma":[0.6575894,0.00000809202,0.3411379,0.0002766967,0.0001490221,0.0003034755,0.00003201601,0.000007059094,0.000496419],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.574823,"threshold_uncertainty_score":0.3536789,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2404719341550069,"score_gpt":0.4683928688355242,"score_spread":0.2279209346805173,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}