{"id":"W309965191","doi":"","title":"Alternative Strategies for Large Scale Student Assessment in Canada: Is Value-Added Assessment One Possible Answer","year":2005,"lang":"en","type":"article","venue":"Canadian Journal of Educational Administration and Policy","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Accountability; Scale (ratio); Popularity; Value (mathematics); Educational assessment; Student achievement; Public relations; Academic achievement; Political science; Psychology; Pedagogy; Computer science; Geography; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001907971,0.0001490597,0.0002694743,0.0006327828,0.0002544207,0.0004744258,0.0003454293,0.00004666285,0.002392065],"category_scores_gemma":[0.0001980515,0.0001314494,0.00006836406,0.0004069912,0.00004585379,0.0007559163,0.00001198562,0.0002105959,0.000006664972],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001818537,"about_ca_system_score_gemma":0.1294992,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.2823935,"about_ca_topic_score_gemma":0.9757282,"domain_scores_codex":[0.9972652,0.0001432893,0.0009986304,0.0002293118,0.00103751,0.0003260911],"domain_scores_gemma":[0.997339,0.0004858905,0.0005303138,0.000170602,0.0008721894,0.0006020711],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00004413803,0.0004564645,0.3283936,0.00001905531,0.00009760114,0.000004171309,0.007650571,0.003027669,0.00005591802,0.6166084,0.03575521,0.007887216],"study_design_scores_gemma":[0.001213384,0.0002951501,0.8926193,0.0000438826,0.0000194258,0.00002991461,0.01147773,0.005517395,0.0000794597,0.03240937,0.05609335,0.0002016532],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7936091,0.0001528049,0.001454403,0.189733,0.0006396387,0.0004153339,0.0003120107,0.000001463035,0.01368222],"genre_scores_gemma":[0.9881157,0.0000281596,0.004467573,0.005215203,0.0007972459,0.00002810855,0.00002934398,0.000007859873,0.001310783],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6933348,"threshold_uncertainty_score":0.9985199,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1030903987976077,"score_gpt":0.4976632851093496,"score_spread":0.3945728863117419,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}