{"id":"W2255198894","doi":"10.20381/ruor-20086","title":"Implications of the multiple-use of large-scale assessments for the process of validation: A case study of the multiple-use of a Grade 9 mathematics assessment","year":2011,"lang":"en","type":"dissertation","venue":"uO Research (University of Ottawa)","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Accountability; Scale (ratio); Process (computing); Argument (complex analysis); Quality (philosophy); Perspective (graphical); Test (biology); Data collection; Computer science; Management science; Data science; Psychology; Mathematics education; Engineering; Mathematics; Artificial intelligence; Medicine; Statistics; Political science; Geography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002229478,0.0001906869,0.0005857495,0.000302182,0.000883676,0.00002656174,0.001862088,0.0002049234,0.00007124378],"category_scores_gemma":[0.0007088065,0.0001382764,0.0003757765,0.001328978,0.0007884796,0.0003611569,0.000336304,0.0003751783,2.667022e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001067704,"about_ca_system_score_gemma":0.0008969409,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.007124819,"about_ca_topic_score_gemma":0.05770664,"domain_scores_codex":[0.9960122,0.0007163087,0.0006377156,0.0003141407,0.001984295,0.0003353877],"domain_scores_gemma":[0.991384,0.002881114,0.001828754,0.0009550102,0.002882978,0.00006808908],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.00008096181,0.007099147,0.8171542,0.001530145,0.0008856162,0.000002374316,0.1611869,0.0000825182,0.002131137,0.008817037,0.0005400478,0.0004899783],"study_design_scores_gemma":[0.001587435,0.0003318813,0.4513003,0.000262316,0.0004663646,7.639726e-7,0.5429348,0.000847506,0.00143235,0.0003437722,0.0003590481,0.0001334712],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9927964,0.00001082905,0.0003903233,0.0001736599,0.00011792,0.004671287,0.0005128722,0.000007695098,0.001319043],"genre_scores_gemma":[0.9950502,0.00006538533,0.001981968,0.000001639674,0.00001599235,0.00003268978,0.00004637973,0.00001998907,0.002785791],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.381748,"threshold_uncertainty_score":0.9994868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1975998779485343,"score_gpt":0.4601881434004951,"score_spread":0.2625882654519608,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}