{"id":"W2255198894","doi":"10.20381/ruor-20086","title":"Implications of the multiple-use of large-scale assessments for the process of validation: A case study of the multiple-use of a Grade 9 mathematics assessment","year":2011,"lang":"en","type":"dissertation","venue":"uO Research (University of Ottawa)","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Accountability; Scale (ratio); Process (computing); Argument (complex analysis); Quality (philosophy); Perspective (graphical); Test (biology); Data collection; Computer science; Management science; Data science; Psychology; Mathematics education; Engineering; Mathematics; Artificial intelligence; Medicine; Statistics; Political science; Geography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2171568,0.00139252,0.001591255,0.00492502,0.03383186,0.01967397,0.008038285,0.01078385,0.001745885],"category_scores_gemma":[0.3146715,0.002659301,0.002187129,0.005786421,0.04346599,0.02766652,0.02636309,0.01518751,0.0006362509],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01972363,"about_ca_system_score_gemma":0.02219641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01779079,"about_ca_topic_score_gemma":0.02587624,"domain_scores_codex":[0.5047944,0.4394453,0.01073454,0.007644144,0.02881451,0.008567005],"domain_scores_gemma":[0.4588274,0.4674942,0.01978603,0.02566791,0.02219485,0.006029563],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"case_report","study_design_scores_codex":[0.00006499186,0.000336624,0.01643484,0.0002319665,0.00003514058,0.006756857,0.9163907,0.0008504001,0.0008209302,0.03799329,0.001097242,0.01898714],"study_design_scores_gemma":[0.00008030575,0.0005131589,0.0105917,0.001660477,0.00005867421,0.006468888,0.8958071,0.005441823,0.002937662,0.0282037,0.04800296,0.0002334722],"study_design_candidate":"case_report","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8585783,0.0009373532,0.07135868,0.02412385,0.000170178,0.001388098,0.00006683246,0.0001423163,0.04323443],"genre_scores_gemma":[0.9679044,0.0003471048,0.02685333,0.001310978,0.00004590795,0.001011269,0.00002497946,0.0001028389,0.002399211],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2171568,"threshold_uncertainty_score":0.965385,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1975998779485343,"score_gpt":0.4601881434004951,"score_spread":0.2625882654519608,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}