{"id":"W3025093973","doi":"10.1007/s10833-020-09380-5","title":"Large-scale assessments and their effects: The case of mid-stakes tests in Ontario","year":2020,"lang":"en","type":"article","venue":"Journal of Educational Change","topic":"Global Educational Reforms and Inequalities","field":"Social Sciences","cited_by":37,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Accountability; Excellence; Incentive; Equity (law); Transparency (behavior); Psychology; Standardized test; Scale (ratio); Pace; Test (biology); Literacy; Political science; Pedagogy; Economics; Mathematics education; Law; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02434277,0.0004917139,0.0007700205,0.001648091,0.005582884,0.003234338,0.002695362,0.002150337,0.004229585],"category_scores_gemma":[0.1359895,0.0004525964,0.0005048648,0.002951922,0.003727785,0.003469442,0.003476502,0.002686409,0.0004445524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.03448345,"about_ca_system_score_gemma":0.0549152,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9272757,"about_ca_topic_score_gemma":0.9610855,"domain_scores_codex":[0.9794901,0.01146153,0.0006432728,0.001028784,0.004692465,0.002683889],"domain_scores_gemma":[0.7863542,0.1515372,0.009538953,0.01007083,0.03164849,0.01085036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002709221,0.003177702,0.5907605,0.0005946918,0.0003527233,0.002108318,0.03501081,0.009917397,0.001537009,0.02288365,0.02811014,0.3028378],"study_design_scores_gemma":[0.0004975996,0.001081841,0.9129084,0.0005410575,0.0002590265,0.000216503,0.02893975,0.01391244,0.001463553,0.01141443,0.02852662,0.0002386412],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9434316,0.001050582,0.002387196,0.01029747,0.00008914495,0.0006479148,0.0007813266,0.0001660711,0.04114866],"genre_scores_gemma":[0.9953073,0.0002054296,0.001405477,0.0002860691,0.00002207506,0.0000995102,0.00008374678,0.00001214679,0.002578381],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07272428,"threshold_uncertainty_score":0.2501961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0907132103583014,"score_gpt":0.3791114910045938,"score_spread":0.2883982806462925,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}