{"id":"W4251721468","doi":"10.1177/0265532220925448","title":"Understanding writing quality change: A longitudinal study of repeaters of a high-stakes standardized English proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Language proficiency; Psychology; Test (biology); Sophistication; Linguistics; Mathematics education","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001772558,0.0003104826,0.0004894471,0.0009956155,0.0008225604,0.0007769205,0.0006696213,0.0008115863,0.0008167771],"category_scores_gemma":[0.008109648,0.0003281676,0.0005305953,0.0006929956,0.0004082769,0.0007759631,0.0006273013,0.0009766462,0.0004680221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005514867,"about_ca_system_score_gemma":0.0004344385,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01896972,"about_ca_topic_score_gemma":0.02083797,"domain_scores_codex":[0.9989138,0.0002186481,0.0000995997,0.0001801501,0.0004119833,0.0001758825],"domain_scores_gemma":[0.9937552,0.0008144168,0.002485103,0.0005324286,0.001644813,0.0007680209],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00007691726,0.0001943209,0.9940573,0.000006344445,0.00003903639,0.0001603362,0.001693944,0.00002774115,0.0004612088,0.000009713145,0.00006621791,0.003206789],"study_design_scores_gemma":[0.00000198218,0.0002879178,0.9984547,0.000002839126,0.000009066498,0.0001885287,0.0007001489,0.00009094222,0.0001181114,0.000007958578,0.0001323055,0.000005528087],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9996734,0.00004487671,0.00005598756,0.00002000555,0.000002046551,0.000007708416,0.00006059833,0.000002364795,0.0001330857],"genre_scores_gemma":[0.9993008,0.00002569825,0.00007934061,0.00001217192,0.000002639247,0.000009019112,0.0001659085,0.000002683144,0.0004017013],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01896972,"threshold_uncertainty_score":0.03771859,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4041793967710166,"score_gpt":0.4111835522633015,"score_spread":0.007004155492284891,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}