{"id":"W4392858454","doi":"10.1145/3626253.3635537","title":"Alternative Evaluation in CS Education Research: A Systematic Literature Map","year":2024,"lang":"en","type":"article","venue":"","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Grading (engineering); Computer science; Variety (cybernetics); Mathematics education; Artificial intelligence; Psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1232779,0.001439487,0.00653894,0.05998795,0.002916407,0.009344759,0.002184204,0.003142365,0.003076073],"category_scores_gemma":[0.2929019,0.001826966,0.005490649,0.04868529,0.005479698,0.009883196,0.007121698,0.003571288,0.0004375919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01342748,"about_ca_system_score_gemma":0.04234836,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01203094,"about_ca_topic_score_gemma":0.03660174,"domain_scores_codex":[0.8682486,0.06467696,0.04180326,0.005606309,0.01839805,0.001266932],"domain_scores_gemma":[0.498782,0.4266817,0.02610083,0.008802443,0.03751129,0.002121635],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0003033721,0.0001195721,0.005273886,0.5487493,0.003310923,0.0002257123,0.003761398,0.000446314,0.0003882735,0.006526503,0.004262311,0.4266324],"study_design_scores_gemma":[0.00008852953,0.0001518218,0.003377815,0.9439735,0.005510029,0.0002675653,0.002808249,0.0003565934,0.0002404959,0.00501626,0.0381412,0.00006792432],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.002497213,0.9895571,0.003001842,0.002194945,0.0001978303,0.0008975961,0.0003851878,0.00002768729,0.001240544],"genre_scores_gemma":[0.04680282,0.9297666,0.01814043,0.001784339,0.0001900567,0.00252537,0.0005541222,0.00004526819,0.0001910258],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.876722,"threshold_uncertainty_score":0.6519638,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07106789224654667,"score_gpt":0.4350405050365657,"score_spread":0.363972612790019,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}