{"id":"W4224251276","doi":"10.5539/ijel.v12n3p99","title":"Evaluating and Testing English Language Skills: Benchmarking the TOEFL and IELTS Tests","year":2022,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Educational Practices and Challenges","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Construct validity; Psychology; Mathematics education; Test (biology); Rasch model; Active listening; Validity; Face validity; Language assessment; Reliability (semiconductor); Reading comprehension; Reading (process); Medical education; Psychometrics; Clinical psychology; Linguistics; Developmental psychology; Medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01007659,0.000566847,0.000424602,0.003238196,0.0004086586,0.001547211,0.0007492704,0.0006107,0.00143879],"category_scores_gemma":[0.02828799,0.0001477894,0.0006894713,0.001814909,0.000583226,0.001613258,0.001637276,0.0006079936,0.0006474087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008791345,"about_ca_system_score_gemma":0.001792982,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003570443,"about_ca_topic_score_gemma":0.00577103,"domain_scores_codex":[0.9868162,0.004322933,0.001850472,0.0008166618,0.00560129,0.0005924214],"domain_scores_gemma":[0.9797385,0.005990608,0.00303825,0.0007963203,0.009306311,0.001130011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006158523,0.001446549,0.6488209,0.0005993699,0.0001848766,0.0004294339,0.004756558,0.00306279,0.006378973,0.001287516,0.002327926,0.3300892],"study_design_scores_gemma":[0.00004352932,0.003313285,0.9501462,0.0004365292,0.0001279469,0.0008643768,0.008719571,0.008104893,0.01414074,0.0009557045,0.01304844,0.00009883452],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9787363,0.0007805245,0.008119093,0.0002213014,0.00007056271,0.0003842992,0.0004894182,0.000121216,0.01107734],"genre_scores_gemma":[0.9801776,0.0004605213,0.01534593,0.00006722919,0.00002072543,0.0002442016,0.0008677103,0.00002945149,0.002786638],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01007659,"threshold_uncertainty_score":0.05329078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05794822096153737,"score_gpt":0.4157850226435775,"score_spread":0.3578368016820401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}