{"id":"W4382501122","doi":"10.5539/elt.v16n6p1","title":"A Comparative-Correlative Study of Test Rubrics Used as Benchmarks in Assessing IELTS and TOEFL Speaking Skills","year":2023,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Rubric; Psychology; Test (biology); Mathematics education; Correlation; Language assessment; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03074645,0.0003213342,0.0003925798,0.004127366,0.0008530416,0.0008123016,0.0005373553,0.0003625531,0.001191061],"category_scores_gemma":[0.0990337,0.0002703093,0.000532104,0.002879195,0.00094102,0.0009666148,0.001341022,0.0004890903,0.0004798357],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001125028,"about_ca_system_score_gemma":0.001030359,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001807144,"about_ca_topic_score_gemma":0.003585788,"domain_scores_codex":[0.9669516,0.02272912,0.002361551,0.001818744,0.005497877,0.0006411203],"domain_scores_gemma":[0.8718928,0.08159952,0.009384012,0.00602943,0.02981074,0.001283557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001825346,0.001588543,0.8361986,0.0004559338,0.0002812511,0.0003326431,0.02088998,0.0004045941,0.006904854,0.001219223,0.0009336896,0.1289654],"study_design_scores_gemma":[0.00008242491,0.008239741,0.9699692,0.0001537365,0.0001537118,0.0007297911,0.009150485,0.001228948,0.007053553,0.0002121346,0.00297626,0.00005000502],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.992347,0.0003913102,0.003760091,0.00002480108,0.00004482838,0.0002618006,0.0001140644,0.00002014084,0.003036005],"genre_scores_gemma":[0.9958462,0.0001190017,0.003130986,0.00002049466,0.00002275265,0.0002566131,0.0001704081,0.00001141346,0.0004220413],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03074645,"threshold_uncertainty_score":0.1626047,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01712188340033934,"score_gpt":0.3452414626050681,"score_spread":0.3281195792047287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}