{"id":"W4415966001","doi":"10.1145/3769994.3770057","title":"Alternative Grading at Scale: Insights from Implementing Weekly Checkpoint Quizzes in a Large Introductory CS Course","year":2025,"lang":"","type":"article","venue":"","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Grading (engineering); Mathematical proof; Course (navigation); Computer aided instruction; Frequently asked questions","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01537753,0.0007615758,0.0005284068,0.0009925611,0.001129183,0.003098559,0.002545984,0.001575001,0.002408654],"category_scores_gemma":[0.133256,0.0004876016,0.0003243558,0.000811649,0.001046958,0.002670785,0.001575637,0.002485007,0.0007841842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001887688,"about_ca_system_score_gemma":0.001963604,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006079669,"about_ca_topic_score_gemma":0.01092837,"domain_scores_codex":[0.9822417,0.01178709,0.0006237,0.001294633,0.002917888,0.001134789],"domain_scores_gemma":[0.8898898,0.07767402,0.006626663,0.009613504,0.01138245,0.004813562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.004478236,0.03191353,0.1637283,0.0008023727,0.0002373331,0.001365845,0.0447156,0.02799779,0.05749914,0.006433337,0.009246345,0.6515821],"study_design_scores_gemma":[0.002691569,0.04707227,0.4816166,0.0005657589,0.0004735047,0.001028901,0.04846234,0.3071344,0.05314146,0.02139485,0.03553135,0.0008869556],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9708369,0.00004708452,0.0228856,0.0005522927,0.00003628831,0.0006981284,0.00006577573,0.0005238273,0.004353959],"genre_scores_gemma":[0.9720021,0.00002156967,0.02586333,0.0002012522,0.000007409534,0.0002845675,0.00007863199,0.00008852787,0.001452585],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01537753,"threshold_uncertainty_score":0.08132517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01362028844344251,"score_gpt":0.2915411522685377,"score_spread":0.2779208638250952,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}