{"id":"W1971125305","doi":"10.1002/j.2333-8504.2003.tb01910.x","title":"INVESTIGATING THE VALIDITY OF TOEFL: A FEASIBILITY STUDY USING CONTENT AND CRITERION‐RELATED STRATEGIES","year":2003,"lang":"en","type":"article","venue":"ETS Research Report Series","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Context (archaeology); Psychology; Test (biology); Content validity; Language proficiency; Test validity; Sample (material); Computer science; Mathematics education; Language assessment; Natural language processing; Psychometrics; Clinical psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1324398,0.0006538657,0.0006865063,0.002597116,0.001293587,0.001841331,0.001857129,0.001412663,0.002032016],"category_scores_gemma":[0.3043944,0.0007800232,0.001091006,0.001103101,0.001897815,0.002959203,0.002967306,0.001475026,0.0005935896],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001590226,"about_ca_system_score_gemma":0.005412686,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001314146,"about_ca_topic_score_gemma":0.002138939,"domain_scores_codex":[0.9174305,0.06027504,0.006416048,0.002807276,0.0113193,0.001751839],"domain_scores_gemma":[0.5955468,0.3115395,0.01460939,0.01721021,0.05806443,0.00302969],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.007634807,0.02805565,0.548812,0.001220358,0.0002737277,0.0007246249,0.01945046,0.004139605,0.02259444,0.006816443,0.001595434,0.3586825],"study_design_scores_gemma":[0.007019191,0.1012731,0.676852,0.001246881,0.0005521363,0.001585854,0.02681949,0.1023245,0.05626528,0.006242616,0.01940056,0.0004183181],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9566503,0.00005950067,0.02751952,0.0004061546,0.00008037399,0.01074959,0.0001354533,0.00009319208,0.004305913],"genre_scores_gemma":[0.9079462,0.00006231301,0.07577714,0.0002993807,0.00005027604,0.01482402,0.0001807143,0.00004567076,0.000814369],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1324398,"threshold_uncertainty_score":0.7004169,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5438925237843821,"score_gpt":0.4462797001089897,"score_spread":0.09761282367539248,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}