{"id":"W2019289429","doi":"10.1177/082957350001600105","title":"A Critical Analysis of Grade 3 Testing in Ontario","year":2000,"lang":"en","type":"article","venue":"Canadian Journal of School Psychology","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Curriculum; Psychology; Cognition; Cognitive test; Standardized test; Medical education; Educational assessment; Test (biology); Applied psychology; Mathematics education; Pedagogy; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004136708,0.0002963792,0.0003342251,0.005349661,0.007558038,0.003391506,0.001121473,0.0005416746,0.002468196],"category_scores_gemma":[0.03008349,0.0003208986,0.0004115257,0.008337241,0.002551083,0.0007211537,0.002092997,0.0007314107,0.000206213],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.08028977,"about_ca_system_score_gemma":0.06458399,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9412885,"about_ca_topic_score_gemma":0.9694316,"domain_scores_codex":[0.9944087,0.0007683266,0.0004337373,0.0003415374,0.002803913,0.001243857],"domain_scores_gemma":[0.9568312,0.006652301,0.005090039,0.0008876165,0.02811967,0.002419123],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.0004325524,0.00008994628,0.3588598,0.0005834633,0.00004137862,0.003788807,0.4528161,0.0003988718,0.003334428,0.01169667,0.02464324,0.1433147],"study_design_scores_gemma":[0.000004640357,0.00006785864,0.8034925,0.00026487,0.00002318069,0.0001440634,0.1408413,0.0002003194,0.000930239,0.0005665777,0.05341658,0.00004790236],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.929956,0.001161003,0.000809359,0.002935117,0.0001191307,0.0005640583,0.001738206,0.00006718859,0.06265002],"genre_scores_gemma":[0.9846297,0.0006720625,0.0009027114,0.0003354645,0.00002040889,0.0001840622,0.0006347926,0.0000278869,0.01259289],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08028977,"threshold_uncertainty_score":0.5825458,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1438827238597663,"score_gpt":0.4307582175762912,"score_spread":0.286875493716525,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}