{"id":"W1971125305","doi":"10.1002/j.2333-8504.2003.tb01910.x","title":"INVESTIGATING THE VALIDITY OF TOEFL: A FEASIBILITY STUDY USING CONTENT AND CRITERION‐RELATED STRATEGIES","year":2003,"lang":"en","type":"article","venue":"ETS Research Report Series","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Context (archaeology); Psychology; Test (biology); Content validity; Language proficiency; Test validity; Sample (material); Computer science; Mathematics education; Language assessment; Natural language processing; Psychometrics; Clinical psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.007013302,0.0001298127,0.0002445516,0.00008450144,0.001573997,0.0005526277,0.0001264494,0.00003764588,0.0002595421],"category_scores_gemma":[0.003022952,0.00008642272,0.00004898327,0.00008944524,0.001376146,0.0004868943,0.0001114993,0.0006635842,0.000001660367],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004090185,"about_ca_system_score_gemma":0.0002171462,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002599136,"about_ca_topic_score_gemma":0.0009764915,"domain_scores_codex":[0.9966228,0.001699173,0.0004985254,0.0002989809,0.0005765628,0.0003039604],"domain_scores_gemma":[0.9985308,0.0003407855,0.0001933455,0.0003955294,0.0004635819,0.00007593726],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00004020946,0.0005395424,0.3158163,0.0003090967,0.0001961901,0.0002988027,0.5237395,0.00006486661,0.005306699,0.1528761,0.0002153473,0.0005973483],"study_design_scores_gemma":[0.0005570443,0.001230132,0.07803874,0.0002380291,0.00004806123,0.0003857601,0.8791475,0.0001200095,0.0008419356,0.02351139,0.01555473,0.0003266314],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9841573,0.0002134551,0.000005046367,0.0002421663,0.0001366207,0.0004641621,0.000002801223,0.00005115774,0.01472728],"genre_scores_gemma":[0.9971981,0.000008505272,0.000104216,0.000009125568,0.00005503007,0.00001248074,0.000003175457,0.00001655578,0.002592792],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.355408,"threshold_uncertainty_score":0.9997258,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5438925237843821,"score_gpt":0.4462797001089897,"score_spread":0.09761282367539248,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}