{"id":"W4382501122","doi":"10.5539/elt.v16n6p1","title":"A Comparative-Correlative Study of Test Rubrics Used as Benchmarks in Assessing IELTS and TOEFL Speaking Skills","year":2023,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Rubric; Psychology; Test (biology); Mathematics education; Correlation; Language assessment; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009728743,0.0001319305,0.000243087,0.0003323914,0.0001489011,0.0001070753,0.0003781028,0.00006373529,0.000008669044],"category_scores_gemma":[0.0009023122,0.0001278709,0.00002160772,0.0006156547,0.00003913616,0.0006255502,0.0002960757,0.0005764795,0.000003209598],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005254414,"about_ca_system_score_gemma":0.00005966008,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002132735,"about_ca_topic_score_gemma":0.0001461262,"domain_scores_codex":[0.9987327,0.0002184947,0.0002632987,0.0003381743,0.0002338822,0.000213482],"domain_scores_gemma":[0.998079,0.001419493,0.0001459071,0.0002847446,0.00003335754,0.00003751128],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.000001056958,0.0007359122,0.1978876,0.00001270927,0.00002256586,0.00007565095,0.7865847,0.0002353892,0.0008237774,0.006412683,0.00004299409,0.007164937],"study_design_scores_gemma":[0.001081339,0.0002956825,0.3675767,0.0002799403,0.00001456847,0.000008814515,0.6150075,0.01290648,0.00105117,0.001377956,0.0000545328,0.0003453082],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9918439,0.00006345625,0.00148473,0.00008773012,0.0002267127,0.000224558,0.000001253664,0.0001766244,0.005891081],"genre_scores_gemma":[0.9945827,0.000001533883,0.005240696,0.00002335134,0.00005032353,0.0000171012,0.000008619742,0.000006538512,0.00006914488],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1715772,"threshold_uncertainty_score":0.5214425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01712188340033934,"score_gpt":0.3452414626050681,"score_spread":0.3281195792047287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}