{"id":"W4407557145","doi":"","title":"Comparative judgement and its impact on the quality of students' written work in mathematics","year":2024,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Okanagan University College; University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Judgement; Work (physics); Quality (philosophy); Mathematics education; Computer science; Mathematics; Engineering; Epistemology; Mechanical engineering; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02927493,0.0002950842,0.0008932614,0.003219505,0.00196172,0.005699097,0.001332429,0.001348198,0.01420545],"category_scores_gemma":[0.3793295,0.0002286228,0.0005434143,0.003583333,0.003513194,0.002986128,0.004160888,0.001538836,0.001020708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002657262,"about_ca_system_score_gemma":0.002541942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003571337,"about_ca_topic_score_gemma":0.0033972,"domain_scores_codex":[0.9523802,0.03108817,0.00260626,0.002712912,0.01025521,0.0009572922],"domain_scores_gemma":[0.3622918,0.5750931,0.01826837,0.009612322,0.02713211,0.007602436],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.02013058,0.003024652,0.4408724,0.002406289,0.0006431509,0.001939893,0.1378956,0.004139472,0.00823526,0.02276012,0.005968411,0.3519841],"study_design_scores_gemma":[0.0006362782,0.004608577,0.8768218,0.001090582,0.0004785542,0.001158347,0.05541634,0.006683376,0.01008545,0.03015095,0.01263055,0.0002391987],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9582391,0.00125027,0.001635389,0.001047703,0.0001971312,0.00006292896,0.0001524256,0.00003160325,0.03738344],"genre_scores_gemma":[0.9977911,0.00009214004,0.0003604509,0.00002804512,0.00002785432,0.00001548293,0.00003673648,0.00001317156,0.001635162],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02927493,"threshold_uncertainty_score":0.1548225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08000636061367551,"score_gpt":0.4273242405371565,"score_spread":0.347317879923481,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}