{"id":"W4307159924","doi":"10.1002/ets2.12357","title":"Mapping<i>TOEFL</i>®<i>Essentials</i>™ Test Scores to the Canadian Language Benchmarks","year":2022,"lang":"en","type":"article","venue":"ETS Research Report Series","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Test (biology); Construct (python library); Psychology; Mathematics education; Computer science; Language assessment; English language; Test score; Natural language processing; Standardized test; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006203053,0.0006928912,0.0004103221,0.006064069,0.00146372,0.001636707,0.001434248,0.0003400186,0.002814326],"category_scores_gemma":[0.03342884,0.0002130445,0.0007281815,0.004453022,0.001083042,0.0008149349,0.001936032,0.00112132,0.0005722762],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01282817,"about_ca_system_score_gemma":0.03328196,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.8693554,"about_ca_topic_score_gemma":0.9032823,"domain_scores_codex":[0.9918219,0.001068808,0.0003616792,0.0005392403,0.00534089,0.0008674165],"domain_scores_gemma":[0.9832523,0.002560048,0.001568166,0.0006731717,0.01115269,0.000793619],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002182871,0.0003683042,0.5553228,0.0004400772,0.0001172952,0.000172384,0.01189254,0.001979344,0.002312451,0.005459442,0.01208204,0.4096349],"study_design_scores_gemma":[0.00002487964,0.0003185618,0.9695446,0.0001320799,0.00004451264,0.0001096855,0.006974478,0.002194939,0.003615276,0.0006565702,0.01628688,0.00009753574],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8887954,0.0004348361,0.02007563,0.0009001899,0.0001938336,0.002608371,0.007805657,0.0003675887,0.07881845],"genre_scores_gemma":[0.9616278,0.0003311472,0.02703123,0.0001376054,0.00002006322,0.001458905,0.003449758,0.00008457853,0.005858989],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1306446,"threshold_uncertainty_score":0.262828,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09859811587093853,"score_gpt":0.4617409910682972,"score_spread":0.3631428751973587,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}