{"id":"W2751665091","doi":"10.1177/0265532217725776","title":"Developing and evaluating a computerized adaptive testing version of the Word Part Levels Test","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":45,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Affix; Test (biology); Vocabulary; Computerized adaptive testing; Natural language processing; Computer science; Strengths and weaknesses; Word (group theory); Vocabulary development; Psychology; Artificial intelligence; Linguistics; Psychometrics; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01608024,0.0007735039,0.0007038628,0.001609601,0.0005102998,0.001256546,0.001349946,0.0009056579,0.001399216],"category_scores_gemma":[0.03725248,0.0004235517,0.0007813913,0.001090033,0.0007604887,0.001498158,0.001282134,0.001183652,0.0006376651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001336351,"about_ca_system_score_gemma":0.003239521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007416,"about_ca_topic_score_gemma":0.009491154,"domain_scores_codex":[0.9883777,0.004914818,0.001767643,0.001162992,0.003386562,0.000390299],"domain_scores_gemma":[0.9744304,0.01279125,0.001561987,0.00140821,0.008850729,0.0009573869],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002838269,0.01271909,0.5395364,0.000484187,0.0002866397,0.0005901551,0.005836051,0.005216554,0.016838,0.001023015,0.003366538,0.4112651],"study_design_scores_gemma":[0.001338274,0.02663551,0.9034678,0.000192812,0.0004755202,0.001903587,0.002600618,0.02110178,0.02901322,0.000924358,0.01212349,0.0002231711],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9786598,0.0002119476,0.01072461,0.0001527392,0.00008876573,0.006071549,0.0007079515,0.0002529371,0.003129761],"genre_scores_gemma":[0.8737003,0.000488284,0.1097217,0.0002997024,0.00006095397,0.009254976,0.003369177,0.0001129712,0.002991911],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01608024,"threshold_uncertainty_score":0.08504146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1617082178408199,"score_gpt":0.3865890369439115,"score_spread":0.2248808191030916,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}