{"id":"W3178523571","doi":"10.5539/elt.v14n7p95","title":"The Development of STEP, the CEFR-Based English Proficiency Test","year":2021,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Psychology; Active listening; Reliability (semiconductor); Construct validity; Content validity; Language proficiency; Language assessment; Test validity; Validity; Mathematics education; Natural language processing; Computer science; Psychometrics; Communication","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001995463,0.0001745077,0.0001780689,0.00005731434,0.0007212005,0.0001168139,0.0004552518,0.0000891306,0.005230876],"category_scores_gemma":[0.005013435,0.0001094878,0.0001063474,0.000231042,0.00008976771,0.00007119893,0.00008978051,0.0007991936,0.00004357795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004976558,"about_ca_system_score_gemma":0.0001650473,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003291518,"about_ca_topic_score_gemma":0.00006097459,"domain_scores_codex":[0.9978639,0.0006698142,0.0004357315,0.0003184078,0.0003099936,0.0004021825],"domain_scores_gemma":[0.9970914,0.001773136,0.000194447,0.0007107088,0.0001664765,0.00006387577],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00001940899,0.0002619703,0.001498343,0.00002549771,0.00007084657,0.0001036387,0.8974158,0.00002575073,0.00262179,0.003791698,0.00200027,0.09216503],"study_design_scores_gemma":[0.000930112,0.00005139237,0.004668362,0.00007073177,0.0000324602,0.00001032729,0.8744969,0.0001152747,0.007251094,0.000006819421,0.1120798,0.0002867047],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8607506,0.007293323,0.001512775,0.0001890992,0.001474647,0.0002487681,0.00001202748,0.0002495831,0.1282692],"genre_scores_gemma":[0.9925364,6.54629e-7,0.001834248,0.001235889,0.0006921566,0.00004935168,0.00004146253,0.00003351937,0.003576281],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1317859,"threshold_uncertainty_score":0.9956785,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01110321991743183,"score_gpt":0.292937380573598,"score_spread":0.2818341606561662,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}