{"id":"W4323529547","doi":"10.20448/gjelt.v2i1.4128","title":"Investigating the Validity of the IELTS Listening Test","year":2022,"lang":"en","type":"article","venue":"Global Journal of English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Active listening; Test (biology); Construct validity; Popularity; Psychology; Construct (python library); Reading (process); Test validity; Language assessment; Mathematics education; Computer science; Linguistics; Psychometrics; Clinical psychology; Social psychology; Communication","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03768865,0.0003534048,0.0005386745,0.00262081,0.0008706151,0.002024194,0.001196643,0.0008339257,0.001602703],"category_scores_gemma":[0.1313888,0.0002700759,0.0008441665,0.001722813,0.002110488,0.002549857,0.001939479,0.001223251,0.000682455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00133145,"about_ca_system_score_gemma":0.004070373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00336442,"about_ca_topic_score_gemma":0.004828076,"domain_scores_codex":[0.9763752,0.01009345,0.003145675,0.001148605,0.00840225,0.0008348858],"domain_scores_gemma":[0.8946469,0.07053333,0.009856524,0.003034471,0.02030421,0.001624584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003787632,0.0007102487,0.8932775,0.0002666081,0.0001241176,0.0001910672,0.006767523,0.0002558388,0.0009400234,0.002379706,0.000618923,0.09408964],"study_design_scores_gemma":[0.0001085622,0.004339188,0.9630927,0.000868612,0.0002232698,0.0006879102,0.01506153,0.002839797,0.002858649,0.002034201,0.00780358,0.00008199299],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9814714,0.0007288502,0.004021294,0.0009285938,0.0001438802,0.0004632721,0.0002284539,0.00001717338,0.01199708],"genre_scores_gemma":[0.9910586,0.0006248826,0.00510703,0.0003536376,0.00007383177,0.0007339431,0.0004991957,0.00002441065,0.001524398],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03768865,"threshold_uncertainty_score":0.1993191,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01450405785037267,"score_gpt":0.2875674398671974,"score_spread":0.2730633820168247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}