{"id":"W4387267871","doi":"10.1080/02602938.2023.2263668","title":"Flexible assessment: some benefits and costs for students and instructors","year":2023,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Grading (engineering); Workload; Flexibility (engineering); Rigour; Formative assessment; Psychology; Medical education; Mathematics education; Pedagogy; Computer science; Engineering; Management; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02930929,0.000909053,0.000654372,0.001757672,0.003993188,0.007141,0.001892878,0.002754075,0.008429941],"category_scores_gemma":[0.08147968,0.0004633384,0.001327303,0.001563555,0.002083285,0.00713257,0.004848606,0.003326662,0.0009084131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004403525,"about_ca_system_score_gemma":0.006262987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003552384,"about_ca_topic_score_gemma":0.01036312,"domain_scores_codex":[0.9630491,0.01974132,0.002712513,0.0007811934,0.01192514,0.001790791],"domain_scores_gemma":[0.8602651,0.09183064,0.007824074,0.00897613,0.02050302,0.01060112],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0008825122,0.0008954997,0.04127074,0.0007183256,0.00008761794,0.0008221036,0.005728545,0.001082569,0.001863694,0.01540203,0.009237903,0.9220086],"study_design_scores_gemma":[0.0008276778,0.01290331,0.4670411,0.008536581,0.0009727204,0.01193674,0.1362881,0.02215533,0.01359968,0.1237906,0.2006336,0.001314685],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.704805,0.01306226,0.03276627,0.1666551,0.001020796,0.0008277218,0.0001993688,0.0006431419,0.08002033],"genre_scores_gemma":[0.9651903,0.00220171,0.02405697,0.002363957,0.0003562293,0.0003219268,0.00005629963,0.00006792924,0.005384697],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9706907,"threshold_uncertainty_score":0.1550042,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2742143064569399,"score_gpt":0.5825374862958459,"score_spread":0.3083231798389059,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}