{"id":"W4392858454","doi":"10.1145/3626253.3635537","title":"Alternative Evaluation in CS Education Research: A Systematic Literature Map","year":2024,"lang":"en","type":"article","venue":"","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Grading (engineering); Computer science; Variety (cybernetics); Mathematics education; Artificial intelligence; Psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002754536,0.00005911402,0.0000867432,0.0004923103,0.00003787598,0.0006666171,0.0003255658,0.00003773348,0.00001163882],"category_scores_gemma":[0.0003154296,0.00004382571,0.00002746296,0.001230103,0.00001138753,0.000386885,0.00007029924,0.0003043317,0.0002221634],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001388792,"about_ca_system_score_gemma":0.0003551314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003305139,"about_ca_topic_score_gemma":0.000015105,"domain_scores_codex":[0.9983833,0.0004896315,0.0001690571,0.0002541407,0.0005638723,0.0001400251],"domain_scores_gemma":[0.9991556,0.0002226132,0.0000195652,0.0002646116,0.0003029287,0.00003466427],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000149967,0.0001779734,0.00009189325,0.01592177,0.00003043337,0.00004480708,0.01344746,0.0018125,0.0001166915,0.937951,0.005174422,0.02522958],"study_design_scores_gemma":[0.00003704057,0.00002369662,0.00004048941,0.009245146,0.000004048482,0.000007835134,0.0002117301,0.9173651,0.00002670757,0.07269394,0.000293364,0.00005093615],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1173048,0.09384339,0.5091112,0.1728739,0.01273108,0.006422982,0.000008378422,0.002007308,0.08569699],"genre_scores_gemma":[0.9837808,0.00003745358,0.007849453,0.00009321867,0.0001854761,0.00005831169,0.000007787852,0.000005742519,0.007981757],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9155526,"threshold_uncertainty_score":0.6428204,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07106789224654667,"score_gpt":0.4350405050365657,"score_spread":0.363972612790019,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}