{"id":"W3137000688","doi":"10.3138/cjpe.69691","title":"Collaborative Evaluation Designs as an Authentic Course Assessment","year":2021,"lang":"en","type":"article","venue":"Canadian Journal of Program Evaluation","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Authentic assessment; Computer science; Evaluation methods; Graduate students; Authentic learning; Instructional design; Knowledge management; Psychology; Mathematics education; Pedagogy; Curriculum; Multimedia; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"grok","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"opus","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.03691512,0.0001912663,0.0003461285,0.0005725187,0.0004316792,0.001171792,0.0005039414,0.000125141,0.01546253],"category_scores_gemma":[0.006553598,0.0001613679,0.0001479498,0.00183557,0.0001186941,0.001539735,0.00001601005,0.0002956272,0.0001710268],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001115312,"about_ca_system_score_gemma":0.05974879,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000204671,"about_ca_topic_score_gemma":0.02224159,"domain_scores_codex":[0.9884252,0.003615302,0.001288473,0.0003946455,0.005902145,0.0003742486],"domain_scores_gemma":[0.9753858,0.0004339801,0.001096702,0.0005517121,0.02167637,0.0008554197],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001647774,0.0002698736,0.01649594,0.000003769366,0.00009570923,0.00003586505,0.003299156,0.007728069,0.0002514436,0.002511824,0.003426444,0.9658654],"study_design_scores_gemma":[0.004017423,0.002350267,0.2687433,0.0001239755,0.00104838,0.0002574697,0.02479409,0.5616567,0.001018122,0.08269292,0.05276803,0.0005293419],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9690569,0.001460447,0.004543146,0.006147327,0.00344697,0.00264999,0.00002508437,0.00002076482,0.01264938],"genre_scores_gemma":[0.988532,0.00001581473,0.01032383,0.000296206,0.0002611896,0.0001672236,0.00009164624,0.00001640925,0.0002956885],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9653361,"threshold_uncertainty_score":0.9998651,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3648231715123615,"score_gpt":0.5893866504709884,"score_spread":0.2245634789586268,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}