{"id":"W4396625747","doi":"10.3138/cmlr-2023-0022","title":"Examining Performance on an Integrated Writing Task from a Canadian English Language Proficiency Test","year":2024,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Test (biology); Task (project management); Computer science; Natural language processing; Language assessment; Linguistics; Psychology; Mathematics education; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002423977,0.0007447791,0.0004490046,0.002763448,0.001204204,0.001054864,0.0009831865,0.0004214441,0.002402497],"category_scores_gemma":[0.009810943,0.0001898218,0.0003673253,0.003207454,0.0006401501,0.0002983118,0.0008914779,0.0005658463,0.0006598212],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005360674,"about_ca_system_score_gemma":0.01265879,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.7972696,"about_ca_topic_score_gemma":0.9222154,"domain_scores_codex":[0.9971759,0.0002182598,0.0001888557,0.0002674397,0.001894071,0.0002553673],"domain_scores_gemma":[0.9928125,0.0008276431,0.0009215555,0.0001520148,0.004541883,0.0007444275],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003370026,0.0004856944,0.7544212,0.0004970463,0.0001970747,0.0004390097,0.006334492,0.0003514465,0.004612469,0.0004313425,0.005789871,0.2261034],"study_design_scores_gemma":[0.00001222389,0.0001610531,0.9954856,0.00004365059,0.00002846139,0.0001003574,0.000795948,0.0001405588,0.0006347415,0.00003408361,0.002545703,0.00001750813],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9784665,0.001955718,0.001063783,0.0002969614,0.00007737913,0.0003779408,0.001825162,0.00003955647,0.01589706],"genre_scores_gemma":[0.981307,0.002073165,0.002665459,0.0001759015,0.00002633857,0.0002045202,0.002818347,0.00001542796,0.01071382],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2027304,"threshold_uncertainty_score":0.4078487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01757527934679107,"score_gpt":0.2638428090947306,"score_spread":0.2462675297479396,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}