{"id":"W7128079767","doi":"10.1075/task.25007.mic","title":"Validity argument for the use of summative task-based language assessment in a language teaching program for adult immigrants","year":2025,"lang":"en","type":"article","venue":"TASK Journal on Task-Based Language Teaching and Learning","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Summative assessment; Formative assessment; Argument (complex analysis); Process (computing); Literacy; Language assessment; Authentic assessment; Dynamic assessment; Component (thermodynamics); Professional development","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4277923,0.0009143144,0.0009234722,0.003444976,0.007281576,0.008969816,0.004868697,0.00325352,0.001993693],"category_scores_gemma":[0.6683615,0.0009423703,0.001319292,0.001543773,0.01056136,0.007183184,0.01262025,0.004501545,0.0006119736],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01217899,"about_ca_system_score_gemma":0.0206674,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01000121,"about_ca_topic_score_gemma":0.01040119,"domain_scores_codex":[0.5247257,0.4047291,0.01373253,0.009881754,0.04327725,0.003653581],"domain_scores_gemma":[0.1763666,0.6657809,0.0280482,0.05576757,0.07095989,0.003076807],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003409827,0.003032757,0.167014,0.002420878,0.0004699529,0.0007954566,0.3295869,0.008407126,0.009130629,0.1034391,0.009515757,0.3627777],"study_design_scores_gemma":[0.002105734,0.009188372,0.2130601,0.01703588,0.001106704,0.001225734,0.1802407,0.1664922,0.08390424,0.1817938,0.142629,0.001217582],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6382864,0.0006614528,0.2880284,0.01583563,0.000697342,0.00646412,0.0002602062,0.0005799668,0.04918648],"genre_scores_gemma":[0.8919578,0.0000660566,0.1012959,0.001103094,0.00006276532,0.004088121,0.00009665115,0.00008346228,0.001246167],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4277923,"threshold_uncertainty_score":0.705634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03379176575737737,"score_gpt":0.3316846127173966,"score_spread":0.2978928469600192,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}