{"id":"W2551839554","doi":"10.7939/r30c4sk1s","title":"The Comparability of Standardized Paper-and-Pencil and Computer-based Mathematics Tests in Alberta","year":2016,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Comparability; Pencil (optics); Standardized test; Mathematics education; Mathematics; Computer science; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001992499,0.00005055242,0.0001360433,0.00003048216,0.0001491717,0.00001209698,0.0001749175,0.00003725549,0.0002239422],"category_scores_gemma":[0.00005371628,0.00003616897,0.00002730709,0.00008877672,0.0005772908,0.0003202085,0.00006162637,0.00003205404,0.00000144784],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001711846,"about_ca_system_score_gemma":0.0003386171,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.01126947,"about_ca_topic_score_gemma":0.03912094,"domain_scores_codex":[0.9994375,0.0001229269,0.0001074435,0.0001004214,0.000134906,0.00009678515],"domain_scores_gemma":[0.9959872,0.003741313,0.00008642571,0.0001052772,0.00002759675,0.00005222631],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001170114,0.0002019331,0.7569833,0.00005346895,0.00002498803,6.941005e-7,0.02799491,0.000001796744,0.00005659942,0.2093697,0.00175603,0.003439568],"study_design_scores_gemma":[0.003069879,0.0002162728,0.7383988,0.0003248816,0.00004586448,7.546827e-7,0.0190595,0.0004960797,0.0001929486,0.03007527,0.2077838,0.0003359346],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.968401,0.0000347401,0.00005483269,0.01553738,0.00003260101,0.0001184443,0.000004451359,0.000004771137,0.01581181],"genre_scores_gemma":[0.9954542,0.0001267606,0.001693919,0.00003481927,0.00001351832,1.202726e-7,0.0000021589,0.00000247366,0.002672037],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2060278,"threshold_uncertainty_score":0.9953146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02166630646753527,"score_gpt":0.2690205533425785,"score_spread":0.2473542468750433,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}