{"id":"W2992614487","doi":"","title":"Validity Considerations in Designing a Writing Test","year":2015,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Reliability (semiconductor); Psychology; Validity; Test validity; Criterion validity; Test design; Computer science; Construct validity; Test method; Psychometrics; Statistics; Mathematics; Developmental psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2553244,0.001007985,0.001678255,0.005239786,0.003515961,0.005793794,0.00216465,0.002563451,0.002574874],"category_scores_gemma":[0.5289379,0.0009125703,0.00191044,0.002616947,0.006634771,0.00679072,0.003488007,0.002854997,0.00159498],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003017104,"about_ca_system_score_gemma":0.008549741,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00213737,"about_ca_topic_score_gemma":0.002850051,"domain_scores_codex":[0.6671924,0.2217111,0.04236902,0.003925935,0.0623679,0.002433525],"domain_scores_gemma":[0.3874624,0.4526501,0.01602566,0.01679167,0.1244595,0.002610719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00134007,0.0008719189,0.07947525,0.006162142,0.0004727493,0.001287678,0.0340398,0.004804068,0.01423099,0.08008876,0.01513061,0.762096],"study_design_scores_gemma":[0.001249696,0.009561704,0.1678689,0.02937348,0.001753027,0.007414706,0.04931555,0.05314001,0.06137546,0.283999,0.3338464,0.001102092],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1195116,0.006822352,0.7180294,0.03708375,0.003877547,0.02371637,0.0003222706,0.001176126,0.08946058],"genre_scores_gemma":[0.2902552,0.002315045,0.6797766,0.004643044,0.001147533,0.01608022,0.0002989766,0.0004693238,0.00501401],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2553244,"threshold_uncertainty_score":0.9183176,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1300524594513647,"score_gpt":0.3466930779898806,"score_spread":0.2166406185385159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}