{"id":"W2133584145","doi":"","title":"Improving the Utility of Large-Scale Assessments in Canada","year":2014,"lang":"en","type":"article","venue":"Canadian Journal of Education / Revue canadienne de l éducation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Strengths and weaknesses; Scale (ratio); Schedule; Mathematics education; Reliability (semiconductor); Psychology; Computer science; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01390499,0.0005386876,0.0006675302,0.003892792,0.003719432,0.003077951,0.001802996,0.0004772357,0.003329088],"category_scores_gemma":[0.0346649,0.0006220286,0.0004522266,0.005602511,0.0009565805,0.001128897,0.002534259,0.001168572,0.0005300973],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.06414459,"about_ca_system_score_gemma":0.1857271,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9890835,"about_ca_topic_score_gemma":0.9954456,"domain_scores_codex":[0.9889414,0.002826474,0.0008157652,0.0007640929,0.005399573,0.001252761],"domain_scores_gemma":[0.9425539,0.008590992,0.001976487,0.001769119,0.03987757,0.005232017],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0009780379,0.0008355011,0.2147726,0.001182965,0.0001710877,0.0006226124,0.009035869,0.005796193,0.002652273,0.006700957,0.03990151,0.7173505],"study_design_scores_gemma":[0.0002717431,0.0004513204,0.9059396,0.0008389924,0.00008743498,0.0002163454,0.006931858,0.01327449,0.002209952,0.001549265,0.06801017,0.0002187102],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8241224,0.007942061,0.02925876,0.01216117,0.0005597861,0.007034001,0.007853434,0.002530498,0.1085379],"genre_scores_gemma":[0.9528651,0.002477253,0.03036068,0.0004158648,0.00003160701,0.0007456507,0.001549798,0.0001267413,0.01142736],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.986095,"threshold_uncertainty_score":0.4654037,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1160623273922698,"score_gpt":0.4076064290081847,"score_spread":0.2915441016159148,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}