{"id":"W4414298304","doi":"10.1002/tesq.70032","title":"Automated Diagnostic Feedback vs. Self‐Assessment: Rethinking Feedback Mechanisms on Academic Writing Development","year":2025,"lang":"en","type":"article","venue":"TESOL Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Toronto; Educational Testing Service","keywords":"Academic writing; Peer feedback; Vocabulary; Task (project management); Graduate students; Second language writing; Higher education; Corrective feedback","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.002004811,0.0004504556,0.000505634,0.0003638891,0.001910708,0.0004591309,0.0009924336,0.0004613082,0.0003120214],"category_scores_gemma":[0.0001913449,0.0004712942,0.0001443077,0.001047292,0.0001544082,0.0004554807,0.0001106266,0.0008866962,0.0004738582],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009296854,"about_ca_system_score_gemma":0.000911129,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001851856,"about_ca_topic_score_gemma":0.000214805,"domain_scores_codex":[0.9954358,0.000439068,0.0008688765,0.0007666904,0.001246526,0.001243049],"domain_scores_gemma":[0.9971017,0.001776795,0.0003095967,0.0003857328,0.0001987599,0.0002274423],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0001392116,0.001207168,0.06644334,0.0003245063,0.0008138269,0.0001209417,0.2277938,0.00004526407,0.002003102,0.5872651,0.03121086,0.08263291],"study_design_scores_gemma":[0.00988844,0.002368839,0.6922545,0.005311303,0.0007163662,0.000008212278,0.126293,0.00813132,0.001546254,0.1011639,0.04751837,0.004799421],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8091307,0.000266818,0.004098051,0.00916269,0.003280427,0.002790813,0.00001284173,0.007054896,0.1642028],"genre_scores_gemma":[0.9862862,0.0000662896,0.009456133,0.001553128,0.0002899276,0.0001741876,0.00003051413,0.00004530912,0.002098277],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6258112,"threshold_uncertainty_score":0.9997739,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01863139032796424,"score_gpt":0.3371814660702979,"score_spread":0.3185500757423337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}