{"id":"W4414209357","doi":"10.1177/02655322251348685","title":"Advancing language assessment for teaching and learning in the era of the artificial intelligence (AI) revolution: Promises and challenges","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"Japan Society for the Promotion of Science; Waseda University","keywords":"Language assessment; Language proficiency; Language acquisition; Applications of artificial intelligence; Assessment for learning; Teaching method","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0604332,0.001100538,0.001660472,0.003185197,0.002531937,0.01572352,0.004402491,0.006751264,0.01120257],"category_scores_gemma":[0.140077,0.0002989584,0.0004747821,0.001962687,0.01246689,0.03060929,0.011512,0.01056729,0.003034393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006519475,"about_ca_system_score_gemma":0.02125966,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007479151,"about_ca_topic_score_gemma":0.00806505,"domain_scores_codex":[0.9674701,0.01989641,0.001298442,0.00124607,0.008777518,0.001311458],"domain_scores_gemma":[0.7965822,0.1352032,0.006129441,0.007563136,0.03781469,0.01670727],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002563958,0.0006985995,0.01041673,0.0008364965,0.00003727399,0.0001266425,0.003653684,0.001268962,0.001109696,0.1569981,0.03932201,0.7852754],"study_design_scores_gemma":[0.00008860134,0.0006662869,0.007008088,0.002417104,0.00002991052,0.0003990383,0.01094088,0.009736213,0.003162809,0.7752033,0.1901772,0.0001705552],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"review","genre_scores_codex":[0.03415008,0.05911489,0.1031336,0.7349624,0.004024211,0.0001685056,0.0003299619,0.001883289,0.06223312],"genre_scores_gemma":[0.7340036,0.03911021,0.164066,0.03933423,0.006668656,0.0003328106,0.000452124,0.0005935305,0.01543888],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.0604332,"threshold_uncertainty_score":0.3196051,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0365263602611972,"score_gpt":0.3059178929668351,"score_spread":0.2693915327056379,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}