{"id":"W4387267871","doi":"10.1080/02602938.2023.2263668","title":"Flexible assessment: some benefits and costs for students and instructors","year":2023,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Grading (engineering); Workload; Flexibility (engineering); Rigour; Formative assessment; Psychology; Medical education; Mathematics education; Pedagogy; Computer science; Engineering; Management; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008769674,0.0001353887,0.0001551822,0.000406448,0.0005706951,0.0003869269,0.0001841109,0.0001086423,0.0002505668],"category_scores_gemma":[0.0004428992,0.0001524707,0.00002371477,0.0005284335,0.00008570851,0.001645508,0.00006721316,0.0001501915,0.00001175935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008181626,"about_ca_system_score_gemma":0.001197883,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007788041,"about_ca_topic_score_gemma":0.0002575704,"domain_scores_codex":[0.9966854,0.0009370854,0.0003586373,0.0004103668,0.001345868,0.0002626148],"domain_scores_gemma":[0.9982723,0.0008162633,0.0002815712,0.0001722369,0.0003486293,0.0001090118],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000008837269,0.0001981262,0.5815931,0.00002665468,0.00002632648,5.477712e-8,0.00294849,0.0001591607,0.00007255514,0.2917919,0.001591317,0.1215835],"study_design_scores_gemma":[0.0008624863,0.00005516007,0.9669791,0.00007540517,0.00007276206,1.956952e-7,0.003677115,0.000927736,0.000007915244,0.01539889,0.01177665,0.0001665297],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9708639,0.0004178045,0.00001938618,0.01948556,0.002148075,0.001767118,0.000006971615,0.0001172684,0.005173947],"genre_scores_gemma":[0.9927557,0.0005610812,0.002606141,0.0003258073,0.0003404812,0.001025432,0.0001572424,0.00001951313,0.002208599],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.385386,"threshold_uncertainty_score":0.6217576,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2742143064569399,"score_gpt":0.5825374862958459,"score_spread":0.3083231798389059,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}