{"id":"W4414298304","doi":"10.1002/tesq.70032","title":"Automated Diagnostic Feedback vs. Self‐Assessment: Rethinking Feedback Mechanisms on Academic Writing Development","year":2025,"lang":"en","type":"article","venue":"TESOL Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Toronto; Educational Testing Service","keywords":"Academic writing; Peer feedback; Vocabulary; Task (project management); Graduate students; Second language writing; Higher education; Corrective feedback","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01540428,0.0004632668,0.0005601902,0.000764747,0.0003241412,0.0013743,0.0006315509,0.0005399146,0.001768379],"category_scores_gemma":[0.09871417,0.0002511116,0.0002997429,0.0003345732,0.0008033744,0.001242261,0.001250248,0.0005831009,0.0002561407],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005135346,"about_ca_system_score_gemma":0.0008667244,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006146102,"about_ca_topic_score_gemma":0.0006192308,"domain_scores_codex":[0.9858714,0.01002828,0.0006368253,0.0008087624,0.002433213,0.0002215153],"domain_scores_gemma":[0.8471927,0.1343994,0.006031065,0.005506377,0.00544411,0.001426316],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.007035602,0.004161377,0.08256211,0.0007779103,0.0001395135,0.00009613204,0.01165263,0.003869733,0.03091948,0.001166101,0.0005620624,0.8570572],"study_design_scores_gemma":[0.002531681,0.05909616,0.6774127,0.001682502,0.001019049,0.0005700593,0.009371215,0.1063712,0.1212891,0.007784794,0.01238925,0.0004822612],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9861683,0.000166614,0.0116755,0.0001377383,0.00003636174,0.000233142,0.00002869278,0.0001540969,0.001399556],"genre_scores_gemma":[0.9904489,0.00006543863,0.008925374,0.00004270838,0.00001436409,0.0001512071,0.00001394191,0.00001530598,0.0003226949],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01540428,"threshold_uncertainty_score":0.08146662,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01863139032796424,"score_gpt":0.3371814660702979,"score_spread":0.3185500757423337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}