{"id":"W2551839554","doi":"10.7939/r30c4sk1s","title":"The Comparability of Standardized Paper-and-Pencil and Computer-based Mathematics Tests in Alberta","year":2016,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Comparability; Pencil (optics); Standardized test; Mathematics education; Mathematics; Computer science; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.006067768,0.000666012,0.0008887892,0.005761118,0.002152498,0.003413754,0.003702701,0.001062092,0.003903385],"category_scores_gemma":[0.0289385,0.0005550692,0.0008509065,0.007401462,0.00195781,0.001072952,0.003339031,0.000820333,0.0007118349],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0325563,"about_ca_system_score_gemma":0.03353858,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9614941,"about_ca_topic_score_gemma":0.9769095,"domain_scores_codex":[0.9914556,0.001620724,0.0006318102,0.001039969,0.003970482,0.001281369],"domain_scores_gemma":[0.9818798,0.004155856,0.001251615,0.001016964,0.01021534,0.001480419],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006386167,0.0008939343,0.8330642,0.0003274259,0.0005588774,0.0008900417,0.005531362,0.004766859,0.002644565,0.005449292,0.005979904,0.1335074],"study_design_scores_gemma":[0.00009893718,0.0002695878,0.9923108,0.00007119643,0.00009309351,0.00009284986,0.002098625,0.001210199,0.0008365046,0.0003624731,0.002518565,0.00003724659],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9835483,0.00129482,0.000802671,0.0003889845,0.00008269031,0.00008326545,0.001100367,0.00006512882,0.01263366],"genre_scores_gemma":[0.992952,0.0003667124,0.0004721728,0.0001427905,0.00001401471,0.00003091331,0.001487258,0.00002530116,0.004508756],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9939322,"threshold_uncertainty_score":0.2362136,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02166630646753527,"score_gpt":0.2690205533425785,"score_spread":0.2473542468750433,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}