{"id":"W7125351318","doi":"","title":"A Comparative Study of Student Perspectives on Technical Writing Feedback Quality: Evaluating LLMs, SLMs, and Humans in Computer Science Topics","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Division of Undergraduate Education; Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"CLARITY; Technical writing; Perspective (graphical); Preference; Course (navigation); Scalability; Quality (philosophy); Peer feedback","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.005750337,0.0003942089,0.001005494,0.0005003331,0.001331431,0.0003301286,0.001038281,0.0001486515,0.00005581616],"category_scores_gemma":[0.0002660754,0.0003946864,0.0001084503,0.00239401,0.002014536,0.0004547815,0.001059513,0.0006680851,0.00001003013],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006228584,"about_ca_system_score_gemma":0.0004845805,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000780908,"about_ca_topic_score_gemma":0.001813654,"domain_scores_codex":[0.9937808,0.001135461,0.001284654,0.00130071,0.001769087,0.0007293014],"domain_scores_gemma":[0.9973765,0.0009359418,0.0004962091,0.0004987856,0.0005618408,0.0001307203],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00005576789,0.002653449,0.7434552,0.00003168629,0.00008180312,0.000003507133,0.2459283,0.00005861498,0.0006570123,0.00566879,0.00001153466,0.001394323],"study_design_scores_gemma":[0.001424968,0.0009550132,0.5945923,0.0002677475,0.00005916286,1.073228e-7,0.402117,0.0002365041,0.00003817915,0.00008708939,0.0000071208,0.0002148638],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9897833,0.0004764628,0.00007010782,0.0007921614,0.0004705865,0.001875517,0.000002813506,0.00004046159,0.006488625],"genre_scores_gemma":[0.9987895,0.0001349884,0.0004350943,0.0001196159,0.0002555618,0.00004999895,8.020323e-7,0.00001046269,0.0002040051],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1561887,"threshold_uncertainty_score":0.9999687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1714072170295567,"score_gpt":0.5004576099750087,"score_spread":0.3290503929454519,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}