{"id":"W4282931634","doi":"10.5539/elt.v15n7p75","title":"Validation of the Results of Linking Speaking Test of IELTS to China’s Standards of English Language Ability","year":2022,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalizability theory; Psychology; Consistency (knowledge bases); Test (biology); Mathematics education; Developmental psychology; Computer science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02889884,0.0005808383,0.0005042645,0.002696614,0.0008762063,0.00106522,0.001047446,0.0006216001,0.001653104],"category_scores_gemma":[0.08304584,0.0002563364,0.001117383,0.001796128,0.001444633,0.001273594,0.001800421,0.0007401581,0.0009028474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008763165,"about_ca_system_score_gemma":0.001987291,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00595244,"about_ca_topic_score_gemma":0.006948308,"domain_scores_codex":[0.9802604,0.006165055,0.003488739,0.002691934,0.006689529,0.0007042927],"domain_scores_gemma":[0.9193493,0.02253293,0.00680307,0.01238374,0.03725611,0.001674761],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004755083,0.0004499518,0.9081758,0.0002511083,0.0002016875,0.0003098716,0.01231014,0.0006395864,0.01311947,0.000958291,0.001155009,0.06195351],"study_design_scores_gemma":[0.0000475861,0.0009279425,0.9639079,0.0001329745,0.0001346,0.0002540605,0.00572151,0.002750911,0.01925048,0.0005195156,0.006296966,0.00005546247],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9853104,0.0001140743,0.005867259,0.0001503973,0.0001268972,0.0005019445,0.0004983459,0.00007859205,0.007352041],"genre_scores_gemma":[0.9882697,0.00007656233,0.008070457,0.0001107289,0.00003015702,0.0005635782,0.0010861,0.00005738066,0.001735386],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02889884,"threshold_uncertainty_score":0.1528335,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007618265817560001,"score_gpt":0.2864872787797649,"score_spread":0.2788690129622049,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}