{"id":"W4323529547","doi":"10.20448/gjelt.v2i1.4128","title":"Investigating the Validity of the IELTS Listening Test","year":2022,"lang":"en","type":"article","venue":"Global Journal of English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Active listening; Test (biology); Construct validity; Popularity; Psychology; Construct (python library); Reading (process); Test validity; Language assessment; Mathematics education; Computer science; Linguistics; Psychometrics; Clinical psychology; Social psychology; Communication","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002025355,0.00007263314,0.0001139421,0.00003127865,0.0005248811,0.00005327161,0.001784425,0.00002313686,0.000008280135],"category_scores_gemma":[0.003711517,0.00004393928,0.000106294,0.0002968509,0.000066642,0.0002184331,0.0006742598,0.0008705694,2.001583e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000125551,"about_ca_system_score_gemma":0.0001963849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003114893,"about_ca_topic_score_gemma":0.000004400821,"domain_scores_codex":[0.9986034,0.0004051583,0.0003057891,0.00009605238,0.0004525799,0.0001370032],"domain_scores_gemma":[0.9985677,0.0004335103,0.0005268784,0.0003151622,0.0001226756,0.00003409023],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.000005161106,0.0006563785,0.5072044,0.00004143553,0.000161647,0.00007105964,0.1290859,0.006150451,0.004221987,0.2915666,0.01046258,0.05037235],"study_design_scores_gemma":[0.003721853,0.001953987,0.3875388,0.001133915,0.0003818117,0.003898951,0.4307599,0.01158252,0.01366797,0.09595665,0.04782181,0.001581786],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9889364,0.0002861625,0.003288346,0.00214536,0.001567362,0.00006038588,0.000007731631,0.00003213283,0.003676118],"genre_scores_gemma":[0.9901346,9.2936e-7,0.009209465,0.0003727459,0.0002489709,0.000002143878,3.700485e-7,0.000002635885,0.00002812157],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.301674,"threshold_uncertainty_score":0.4443301,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01450405785037267,"score_gpt":0.2875674398671974,"score_spread":0.2730633820168247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}